From 4167fcd39b1f1894d3d0d30b623f1de4c3333f09 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 13:15:57 -0400 Subject: [PATCH 01/76] feat(agent): add Claude 5 profile --- lib/components/fabro-agent/src/config.rs | 5 +- lib/components/fabro-agent/src/lib.rs | 3 +- lib/components/fabro-agent/src/memory.rs | 21 +- lib/components/fabro-agent/src/native_tool.rs | 55 +- .../fabro-agent/src/profiles/claude5.rs | 261 +++++++ .../fabro-agent/src/profiles/claude5_tools.rs | 670 ++++++++++++++++++ .../fabro-agent/src/profiles/mod.rs | 160 ++++- .../src/profiles/prompts/claude5.md.j2 | 80 +++ ...ude5_all_conditionals_prompt_snapshot.snap | 89 +++ ...ests__claude5_default_prompt_snapshot.snap | 71 ++ ...sts__claude5_question_prompt_snapshot.snap | 77 ++ ...ubagents_and_question_prompt_snapshot.snap | 87 +++ ...ts__claude5_subagents_prompt_snapshot.snap | 81 +++ ...b_search_and_question_prompt_snapshot.snap | 79 +++ ..._search_and_subagents_prompt_snapshot.snap | 83 +++ ...s__claude5_web_search_prompt_snapshot.snap | 73 ++ .../fabro-agent/src/question_tools.rs | 280 ++++++++ lib/components/fabro-agent/src/session.rs | 101 ++- lib/components/fabro-agent/src/skills.rs | 60 ++ lib/components/fabro-agent/src/subagent.rs | 281 +++++++- .../fabro-agent/src/todo_runtime.rs | 20 +- lib/components/fabro-agent/src/todo_tools.rs | 29 +- lib/components/fabro-agent/src/tools.rs | 2 +- .../fabro-llm/src/adapter_registry.rs | 7 +- .../fabro-workflow/src/handler/llm/api.rs | 38 +- .../fabro-workflow/src/operations/create.rs | 4 +- .../fabro-workflow/src/pipeline/transform.rs | 4 +- .../fabro-workflow/tests/materialize_run.rs | 2 +- lib/foundation/fabro-model/src/adapter.rs | 6 + lib/foundation/fabro-model/src/catalog.rs | 54 +- .../src/catalog/providers/anthropic.toml | 31 +- .../src/catalog/providers/bedrock.toml | 38 +- .../src/catalog/providers/openrouter.toml | 35 +- 33 files changed, 2795 insertions(+), 92 deletions(-) create mode 100644 lib/components/fabro-agent/src/profiles/claude5.rs create mode 100644 lib/components/fabro-agent/src/profiles/claude5_tools.rs create mode 100644 lib/components/fabro-agent/src/profiles/prompts/claude5.md.j2 create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_all_conditionals_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_default_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap create mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap diff --git a/lib/components/fabro-agent/src/config.rs b/lib/components/fabro-agent/src/config.rs index 51165fd3e..a6c8b5c39 100644 --- a/lib/components/fabro-agent/src/config.rs +++ b/lib/components/fabro-agent/src/config.rs @@ -130,7 +130,7 @@ impl NativeToolOptions { // Matched exhaustively so a new profile kind has to state its answer // rather than silently inheriting the default timeout. let default_command_timeout_ms = match profile_kind { - AgentProfileKind::Anthropic => 120_000, + AgentProfileKind::Anthropic | AgentProfileKind::Claude5 => 120_000, // Matches the 60s foreground default Kimi Code's Bash tool // documents, which is what these models are used to budgeting // against. @@ -333,12 +333,15 @@ mod tests { fn native_tool_options_have_expected_profile_defaults() { let openai = NativeToolOptions::for_profile(AgentProfileKind::OpenAi); let anthropic = NativeToolOptions::for_profile(AgentProfileKind::Anthropic); + let claude5 = NativeToolOptions::for_profile(AgentProfileKind::Claude5); let kimi = NativeToolOptions::for_profile(AgentProfileKind::Kimi); assert_eq!(openai.default_command_timeout_ms, 10_000); assert_eq!(openai.max_command_timeout_ms, 600_000); assert_eq!(anthropic.default_command_timeout_ms, 120_000); assert_eq!(anthropic.max_command_timeout_ms, 600_000); + assert_eq!(claude5.default_command_timeout_ms, 120_000); + assert_eq!(claude5.max_command_timeout_ms, 600_000); assert_eq!(kimi.default_command_timeout_ms, 60_000); assert_eq!(kimi.max_command_timeout_ms, 600_000); } diff --git a/lib/components/fabro-agent/src/lib.rs b/lib/components/fabro-agent/src/lib.rs index e8c8b01dc..738faaf23 100644 --- a/lib/components/fabro-agent/src/lib.rs +++ b/lib/components/fabro-agent/src/lib.rs @@ -50,7 +50,8 @@ pub use loop_detection::detect_loop; pub use memory::{MemoryDocument, discover_memory}; pub use native_tool::{NativeTool, ToolVocabulary}; pub use profiles::{ - AgentProfileBuilder, AnthropicProfile, EnvContext, GeminiProfile, KimiProfile, OpenAiProfile, + AgentProfileBuilder, AnthropicProfile, Claude5Profile, EnvContext, GeminiProfile, KimiProfile, + OpenAiProfile, }; pub use question_tools::{ ANTHROPIC_ASK_USER_QUESTION_TOOL, AgentQuestion, AgentQuestionAnswer, diff --git a/lib/components/fabro-agent/src/memory.rs b/lib/components/fabro-agent/src/memory.rs index 41d607ecc..94a88eb40 100644 --- a/lib/components/fabro-agent/src/memory.rs +++ b/lib/components/fabro-agent/src/memory.rs @@ -31,7 +31,9 @@ pub async fn discover_memory( let directories = build_directory_walk(git_root, working_dir); let candidate_filenames: Vec<&str> = match profile_kind { - AgentProfileKind::Anthropic => vec!["AGENTS.md", "CLAUDE.md"], + AgentProfileKind::Anthropic | AgentProfileKind::Claude5 => { + vec!["AGENTS.md", "CLAUDE.md"] + } AgentProfileKind::OpenAi | AgentProfileKind::Gpt56 => { vec!["AGENTS.md", ".codex/instructions.md"] } @@ -207,6 +209,23 @@ mod tests { assert_eq!(anthropic_docs[0].content, "agents"); assert_eq!(anthropic_docs[1].content, "claude"); + let env: Arc = Arc::new(MockSandbox { + files: files.clone(), + ..Default::default() + }); + let claude5_docs = discover_memory( + env.as_ref(), + "/repo", + "/repo", + AgentProfileKind::Claude5, + &CancellationToken::new(), + ) + .await + .unwrap(); + assert_eq!(claude5_docs.len(), 2); + assert_eq!(claude5_docs[0].content, "agents"); + assert_eq!(claude5_docs[1].content, "claude"); + let env: Arc = Arc::new(MockSandbox { files: files.clone(), ..Default::default() diff --git a/lib/components/fabro-agent/src/native_tool.rs b/lib/components/fabro-agent/src/native_tool.rs index bcedb9c53..83b1890ff 100644 --- a/lib/components/fabro-agent/src/native_tool.rs +++ b/lib/components/fabro-agent/src/native_tool.rs @@ -9,8 +9,8 @@ //! //! A [`NativeTool`] is an identity, not a name. The same tool is expressed //! under different names depending on the [`ToolVocabulary`] a profile speaks: -//! fabro's own names by default, Kimi Code's names for the Kimi profile, and -//! Codex's names for the GPT-5.6 profile. +//! fabro's own names by default, Anthropic's names for Claude 5, Kimi Code's +//! names for the Kimi profile, and Codex's names for the GPT-5.6 profile. //! Permissions, categories, and telemetry resolve any name back to the //! identity, so behavior never depends on which vocabulary is in play. //! @@ -26,6 +26,8 @@ pub enum ToolVocabulary { /// Fabro's own names, and the canonical identity used internally. #[default] Fabro, + /// The names Anthropic's Claude 5 coding harness exposes. + Claude5, /// The names Kimi Code exposes, for models trained against that harness. KimiCode, /// The names Codex exposes, for the GPT-5.6 models trained against it. @@ -57,7 +59,11 @@ pub enum NativeTool { Shell, #[strum(to_string = "web_search", serialize = "WebSearch")] WebSearch, - #[strum(to_string = "web_fetch", serialize = "FetchURL")] + #[strum( + to_string = "web_fetch", + serialize = "FetchURL", + serialize = "WebFetch" + )] WebFetch, #[strum(to_string = "spawn_agent")] SpawnAgent, @@ -67,6 +73,14 @@ pub enum NativeTool { Wait, #[strum(to_string = "close_agent")] CloseAgent, + #[strum(to_string = "Agent")] + ClaudeAgent, + #[strum(to_string = "TaskOutput")] + TaskOutput, + #[strum(to_string = "TaskStop")] + TaskStop, + #[strum(to_string = "SendMessage")] + SendMessage, #[strum(to_string = "use_skill", serialize = "Skill")] UseSkill, #[strum(to_string = "update_plan")] @@ -116,6 +130,16 @@ impl NativeTool { pub fn name(self, vocabulary: ToolVocabulary) -> &'static str { match vocabulary { ToolVocabulary::Fabro => self.canonical_name(), + ToolVocabulary::Claude5 => match self { + Self::ReadFile => "Read", + Self::WriteFile => "Write", + Self::EditFile => "Edit", + Self::Shell => "Bash", + Self::WebSearch => "WebSearch", + Self::WebFetch => "WebFetch", + Self::UseSkill => "Skill", + other => other.canonical_name(), + }, ToolVocabulary::KimiCode => match self { Self::ReadFile => "Read", Self::WriteFile => "Write", @@ -177,9 +201,14 @@ impl NativeTool { } Self::WriteFile | Self::EditFile | Self::ApplyPatch => Some(AgentToolCategory::Write), Self::Shell => Some(AgentToolCategory::Shell), - Self::SpawnAgent | Self::SendInput | Self::Wait | Self::CloseAgent => { - Some(AgentToolCategory::Subagent) - } + Self::SpawnAgent + | Self::SendInput + | Self::Wait + | Self::CloseAgent + | Self::ClaudeAgent + | Self::TaskOutput + | Self::TaskStop + | Self::SendMessage => Some(AgentToolCategory::Subagent), // Uncategorized today. Giving these a category would change the CLI // permission gate, which is a behavior change rather than a // classification cleanup, so they keep their existing answer. @@ -261,6 +290,20 @@ mod tests { ); } + #[test] + fn claude5_vocabulary_uses_anthropic_harness_names() { + assert_eq!(NativeTool::ReadFile.name(ToolVocabulary::Claude5), "Read"); + assert_eq!(NativeTool::Shell.name(ToolVocabulary::Claude5), "Bash"); + assert_eq!( + NativeTool::WebFetch.name(ToolVocabulary::Claude5), + "WebFetch" + ); + assert_eq!( + NativeTool::ClaudeAgent.name(ToolVocabulary::Claude5), + "Agent" + ); + } + #[test] fn codex_vocabulary_renames_only_the_shell() { assert_eq!( diff --git a/lib/components/fabro-agent/src/profiles/claude5.rs b/lib/components/fabro-agent/src/profiles/claude5.rs new file mode 100644 index 000000000..6b7826584 --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/claude5.rs @@ -0,0 +1,261 @@ +//! Profile for Claude Fable 5, Opus 5, and Sonnet 5. + +use std::sync::Arc; + +use fabro_model::{AgentProfileKind, Catalog, ProviderId}; + +use super::EnvContext; +use crate::agent_profile::AgentProfile; +use crate::config::NativeToolOptions; +use crate::native_tool::{NativeTool, ToolVocabulary}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, claude5_tools}; +use crate::sandbox::Sandbox; +use crate::skills::Skill; +use crate::subagent::{SessionFactory, SubAgentSupervisor}; +use crate::todo_runtime::TodoRuntime; +use crate::todo_tools::{ + make_task_create_tool, make_task_get_tool, make_task_list_tool, make_task_update_tool, +}; +use crate::tool_registry::ToolRegistry; +use crate::tools::WebFetchSummarizer; + +const CORE_PROMPT: &str = include_str!("prompts/claude5.md.j2"); + +pub struct Claude5Profile { + base: BaseProfile, +} + +impl Claude5Profile { + #[must_use] + pub fn new(model: impl Into) -> Self { + let options = NativeToolOptions::for_profile(AgentProfileKind::Claude5); + Self::with_native_tools(model, &options, None) + } + + pub(crate) fn with_native_tools( + model: impl Into, + options: &NativeToolOptions, + summarizer: Option, + ) -> Self { + Self::with_native_tools_and_todo_runtime( + model, + options, + summarizer, + Arc::new(TodoRuntime::new()), + ) + } + + pub(crate) fn with_native_tools_and_todo_runtime( + model: impl Into, + options: &NativeToolOptions, + summarizer: Option, + todo_runtime: Arc, + ) -> Self { + let mut registry = ToolRegistry::with_vocabulary(ToolVocabulary::Claude5); + registry.register(claude5_tools::make_read_tool()); + registry.register(claude5_tools::make_write_tool()); + registry.register(claude5_tools::make_edit_tool()); + registry.register(claude5_tools::make_bash_tool(options)); + registry.register(claude5_tools::make_web_fetch_tool(summarizer)); + if let Some(api_key) = &options.secrets.brave_search_api_key { + registry.register(claude5_tools::make_web_search_tool(api_key.clone())); + } + + registry.register(claude5_tools::strict_object_tool(make_task_create_tool( + todo_runtime.clone(), + ))); + registry.register(claude5_tools::strict_object_tool(make_task_update_tool( + todo_runtime.clone(), + ))); + registry.register(claude5_tools::strict_object_tool(make_task_get_tool( + todo_runtime.clone(), + ))); + registry.register(claude5_tools::strict_object_tool(make_task_list_tool( + todo_runtime, + ))); + + Self { + base: BaseProfile { + profile_kind: AgentProfileKind::Claude5, + provider_id: ProviderId::anthropic(), + model: model.into(), + catalog: None, + registry, + }, + } + } + + /// Override the transport provider while retaining Claude 5 harness + /// behavior. + #[must_use] + pub fn with_provider_id(mut self, provider_id: ProviderId) -> Self { + self.base.provider_id = provider_id; + self + } + + #[must_use] + pub fn with_catalog(mut self, catalog: Arc) -> Self { + self.base.catalog = Some(catalog); + self + } +} + +impl AgentProfile for Claude5Profile { + fn profile_kind(&self) -> AgentProfileKind { + self.base.profile_kind + } + + fn provider_id(&self) -> ProviderId { + self.base.provider_id.clone() + } + + fn model(&self) -> &str { + &self.base.model + } + + fn catalog(&self) -> Option<&Catalog> { + self.base.catalog.as_deref() + } + + fn tool_registry(&self) -> &ToolRegistry { + &self.base.registry + } + + fn tool_registry_mut(&mut self) -> &mut ToolRegistry { + &mut self.base.registry + } + + fn build_system_prompt( + &self, + env: &dyn Sandbox, + env_context: &EnvContext, + memory: &[String], + user_instructions: Option<&str>, + skills: &[Skill], + ) -> String { + let template = EmbeddedPrompt::new("claude5.md.j2", CORE_PROMPT) + .with_vocabulary(ToolVocabulary::Claude5) + .with_bool( + "has_agent", + self.base + .registry + .get_native(NativeTool::ClaudeAgent) + .is_some(), + ) + .with_bool( + "has_ask_user_question", + self.base + .registry + .get_native(NativeTool::AskUserQuestion) + .is_some(), + ) + .with_bool( + "has_web_search", + self.base + .registry + .get_native(NativeTool::WebSearch) + .is_some(), + ); + + profiles::assemble_system_prompt( + template, + env, + env_context, + memory, + user_instructions, + skills, + ) + } + + fn register_subagent_tools( + &mut self, + supervisor: SubAgentSupervisor, + session_factory: SessionFactory, + current_depth: usize, + ) { + self.base.registry.register(claude5_tools::make_agent_tool( + supervisor.clone(), + session_factory, + current_depth, + )); + self.base + .registry + .register(claude5_tools::make_task_output_tool(supervisor.clone())); + self.base + .registry + .register(claude5_tools::make_task_stop_tool(supervisor.clone())); + self.base + .registry + .register(claude5_tools::make_send_message_tool(supervisor)); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::subagent::SessionFactory; + use crate::test_support::MockSandbox; + + #[test] + fn profile_identity() { + let profile = Claude5Profile::new("claude-fable-5"); + assert_eq!(profile.profile_kind(), AgentProfileKind::Claude5); + assert_eq!(profile.provider_id(), ProviderId::anthropic()); + assert_eq!(profile.model(), "claude-fable-5"); + } + + #[test] + fn core_tools_match_the_accepted_claude5_surface() { + let profile = Claude5Profile::new("claude-sonnet-5"); + let mut names = profile.tool_registry().names(); + names.sort(); + assert_eq!(names, vec![ + "Bash", + "Edit", + "Read", + "TaskCreate", + "TaskGet", + "TaskList", + "TaskUpdate", + "WebFetch", + "Write", + ]); + assert!(!names.iter().any(|name| name == "Grep" || name == "Glob")); + } + + #[test] + fn root_agent_tools_use_claude_names() { + let mut profile = Claude5Profile::new("claude-opus-5"); + let factory: SessionFactory = Arc::new(|| panic!("unused")); + profile.register_subagent_tools(SubAgentSupervisor::new(3), factory, 0); + + for expected in ["Agent", "TaskOutput", "TaskStop", "SendMessage"] { + assert!( + profile.tool_registry().get(expected).is_some(), + "missing {expected}" + ); + } + for absent in ["spawn_agent", "wait", "close_agent", "send_input"] { + assert!( + profile.tool_registry().get(absent).is_none(), + "found {absent}" + ); + } + } + + #[test] + fn prompt_conditionals_follow_registered_tools() { + let env = MockSandbox::linux(); + let profile = Claude5Profile::new("claude-fable-5"); + let prompt = profile.build_system_prompt(&env, &EnvContext::default(), &[], None, &[]); + assert!(!prompt.contains("# Background agents")); + assert!(!prompt.contains("# Asking the user")); + assert!(!prompt.contains("Use `WebSearch`")); + + let mut profile = profile; + let factory: SessionFactory = Arc::new(|| panic!("unused")); + profile.register_subagent_tools(SubAgentSupervisor::new(3), factory, 0); + let prompt = profile.build_system_prompt(&env, &EnvContext::default(), &[], None, &[]); + assert!(prompt.contains("# Background agents")); + } +} diff --git a/lib/components/fabro-agent/src/profiles/claude5_tools.rs b/lib/components/fabro-agent/src/profiles/claude5_tools.rs new file mode 100644 index 000000000..20f284c2b --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/claude5_tools.rs @@ -0,0 +1,670 @@ +//! Claude 5 harness adapters. +//! +//! Execution stays shared with Fabro wherever the behavior agrees. This module +//! narrows the model-facing schemas and supplies the few lifecycle semantics +//! that differ from Fabro's native tools. + +use std::sync::Arc; +use std::time::Duration; + +use fabro_llm::types::ToolDefinition; +use fabro_util::error as util_error; +use serde_json::Value; +use tokio::time; + +use crate::config::NativeToolOptions; +use crate::error::{Error, InterruptReason}; +use crate::native_tool::NativeTool; +use crate::session::Session; +use crate::subagent::{SessionFactory, SubAgentResult, SubAgentStatus, SubAgentSupervisor}; +use crate::tool_registry::{RegisteredTool, ToolContext, ToolSource}; +use crate::tools::{self, WebFetchSummarizer}; + +fn definition( + tool: NativeTool, + description: impl Into, + parameters: Value, +) -> ToolDefinition { + ToolDefinition { + name: tool.canonical_name().to_string(), + description: description.into(), + parameters, + } +} + +/// Reject unknown top-level fields while retaining a shared executor. +#[must_use] +pub(crate) fn strict_object_tool(mut tool: RegisteredTool) -> RegisteredTool { + let object = tool + .definition + .parameters + .as_object_mut() + .expect("native JSON-schema tools should use an object schema"); + object.insert("additionalProperties".to_string(), Value::Bool(false)); + tool +} + +#[must_use] +pub(crate) fn make_read_tool() -> RegisteredTool { + strict_object_tool(tools::make_read_file_tool()) +} + +#[must_use] +pub(crate) fn make_write_tool() -> RegisteredTool { + strict_object_tool(tools::make_write_file_tool()) +} + +#[must_use] +pub(crate) fn make_edit_tool() -> RegisteredTool { + strict_object_tool(tools::make_edit_file_tool()) +} + +#[must_use] +pub(crate) fn make_bash_tool(options: &NativeToolOptions) -> RegisteredTool { + let default_timeout_ms = options.default_command_timeout_ms; + let max_timeout_ms = options.max_command_timeout_ms; + RegisteredTool { + definition: definition( + NativeTool::Shell, + format!( + "Execute a Bash command in a fresh foreground non-login shell. Use this for \ + searches, git inspection, builds, tests, package managers, and terminal \ + operations. Prefer `rg` for content search and `rg --files` for file discovery. \ + Working-directory and environment changes do not persist between calls. \ + `timeout` is in milliseconds, defaults to {default_timeout_ms}, and is capped at \ + {max_timeout_ms}." + ), + serde_json::json!({ + "type": "object", + "properties": { + "command": { + "type": "string", + "description": "Bash source to evaluate." + }, + "timeout": { + "type": "integer", + "minimum": 0, + "maximum": max_timeout_ms, + "description": format!( + "Maximum runtime in milliseconds (default {default_timeout_ms})." + ) + }, + "description": { + "type": "string", + "description": "Short description of what the command does." + } + }, + "required": ["command"], + "additionalProperties": false + }), + ), + executor: Arc::new(move |args, ctx| { + Box::pin(async move { + let command = tools::required_str(&args, "command")?; + let timeout_ms = args + .get("timeout") + .and_then(Value::as_u64) + .unwrap_or(default_timeout_ms) + .min(max_timeout_ms); + tools::run_shell_command(&ctx, command, timeout_ms, None).await + }) + }), + source: ToolSource::Native, + } +} + +#[must_use] +pub(crate) fn make_web_search_tool(api_key: String) -> RegisteredTool { + let mut tool = tools::make_web_search_tool_with_api_key(api_key); + tool.definition = definition( + NativeTool::WebSearch, + "Search the web when current external information is needed. Returns result titles, URLs, \ + and descriptions; use WebFetch to inspect a specific URL.", + serde_json::json!({ + "type": "object", + "properties": { + "query": { + "type": "string", + "description": "The web search query." + } + }, + "required": ["query"], + "additionalProperties": false + }), + ); + tool +} + +#[must_use] +pub(crate) fn make_web_fetch_tool(summarizer: Option) -> RegisteredTool { + let mut tool = tools::make_web_fetch_tool(summarizer); + tool.definition = definition( + NativeTool::WebFetch, + "Fetch an HTTP or HTTPS URL and answer the supplied prompt from its contents.", + serde_json::json!({ + "type": "object", + "properties": { + "url": { + "type": "string", + "description": "The HTTP or HTTPS URL to fetch." + }, + "prompt": { + "type": "string", + "description": "The question or extraction instruction to apply to the page." + } + }, + "required": ["url", "prompt"], + "additionalProperties": false + }), + ); + tool +} + +fn child_session(session_factory: &SessionFactory, ctx: &ToolContext) -> Session { + let mut session = session_factory(); + if let Some(root) = ctx.root_session_id.as_ref().or(ctx.session_id.as_ref()) { + session.set_root_session_id(root.clone()); + } + session +} + +fn format_agent_result(result: &SubAgentResult) -> String { + format!( + "Agent completed (success: {}, turns: {})\n\n{}", + result.success, result.turns_used, result.output + ) +} + +fn format_error(error: &Error) -> String { + util_error::collect_chain(error).join(": ") +} + +#[must_use] +pub(crate) fn make_agent_tool( + supervisor: SubAgentSupervisor, + session_factory: SessionFactory, + current_depth: usize, +) -> RegisteredTool { + RegisteredTool { + definition: definition( + NativeTool::ClaudeAgent, + "Launch a child agent for an independent task. Agents run in the background by \ + default and notify the parent when they finish. Set run_in_background to false to \ + wait for the result synchronously.", + serde_json::json!({ + "type": "object", + "properties": { + "description": { + "type": "string", + "description": "A short 3-5 word description of the task." + }, + "prompt": { + "type": "string", + "description": "The task for the agent to perform." + }, + "run_in_background": { + "type": "boolean", + "description": "Whether to return immediately (default true)." + } + }, + "required": ["description", "prompt"], + "additionalProperties": false + }), + ), + executor: Arc::new(move |args, ctx| { + let supervisor = supervisor.clone(); + let session_factory = session_factory.clone(); + Box::pin(async move { + let description = tools::required_str(&args, "description")?; + let prompt = tools::required_str(&args, "prompt")?; + let run_in_background = args + .get("run_in_background") + .and_then(Value::as_bool) + .unwrap_or(true); + let session = child_session(&session_factory, &ctx); + + if run_in_background { + let task_id = supervisor + .spawn_with_parent_notification( + session, + prompt.to_string(), + description.to_string(), + current_depth, + ) + .map_err(|error| format_error(&error))?; + Ok(format!( + "Agent started in the background.\n\nTask ID: {task_id}" + )) + } else { + let task_id = supervisor + .spawn(session, prompt.to_string(), current_depth) + .map_err(|error| format_error(&error))?; + match supervisor.wait_with_cancel(&task_id, &ctx.cancel).await { + Ok(result) => Ok(format_agent_result(&result)), + Err(Error::Interrupted(InterruptReason::Cancelled)) => { + Err("Cancelled".to_string()) + } + Err(error) => Err(format_error(&error)), + } + } + }) + }), + source: ToolSource::Native, + } +} + +fn required_bool(args: &Value, key: &str) -> Result { + args.get(key) + .and_then(Value::as_bool) + .ok_or_else(|| format!("Missing required boolean parameter: {key}")) +} + +fn required_u64(args: &Value, key: &str) -> Result { + args.get(key) + .and_then(Value::as_u64) + .ok_or_else(|| format!("Missing required non-negative integer parameter: {key}")) +} + +fn finished_output( + supervisor: &SubAgentSupervisor, + task_id: &str, + result: Result, +) -> Result { + supervisor.suppress_parent_notification(task_id); + match result { + Ok(result) => Ok(format_agent_result(&result)), + Err(error) => Err(format_error(&error)), + } +} + +#[must_use] +pub(crate) fn make_task_output_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { + RegisteredTool { + definition: definition( + NativeTool::TaskOutput, + "Get a background agent's current status or wait for its final output. Automatic \ + completion notifications make ordinary polling unnecessary.", + serde_json::json!({ + "type": "object", + "properties": { + "task_id": { + "type": "string", + "description": "The background agent task ID." + }, + "block": { + "type": "boolean", + "default": true, + "description": "Whether to wait for completion." + }, + "timeout": { + "type": "number", + "minimum": 0, + "maximum": 600_000, + "default": 30000, + "description": "Maximum wait time in milliseconds." + } + }, + "required": ["task_id", "block", "timeout"], + "additionalProperties": false + }), + ), + executor: Arc::new(move |args, ctx| { + let supervisor = supervisor.clone(); + Box::pin(async move { + let task_id = tools::required_str(&args, "task_id")?; + let block = required_bool(&args, "block")?; + let timeout_ms = required_u64(&args, "timeout")?; + if timeout_ms > 600_000 { + return Err("timeout must be between 0 and 600000 milliseconds".to_string()); + } + + match supervisor.status(task_id) { + Some(SubAgentStatus::Finished(result)) => { + return finished_output(&supervisor, task_id, result); + } + Some(SubAgentStatus::Running) if !block => { + return Ok(format!("Agent {task_id} is still running.")); + } + Some(SubAgentStatus::Closing | SubAgentStatus::Closed) => { + return Ok(format!("Agent {task_id} has been stopped.")); + } + None => { + return Err(format!( + "No agent found with id: {task_id} (it was never spawned)" + )); + } + Some(SubAgentStatus::Running) => {} + } + + match time::timeout( + Duration::from_millis(timeout_ms), + supervisor.wait_with_cancel(task_id, &ctx.cancel), + ) + .await + { + Ok(Ok(result)) => { + supervisor.suppress_parent_notification(task_id); + Ok(format_agent_result(&result)) + } + Ok(Err(Error::Interrupted(InterruptReason::Cancelled))) => { + supervisor.suppress_parent_notification(task_id); + Err("Cancelled".to_string()) + } + Ok(Err(error)) => { + supervisor.suppress_parent_notification(task_id); + Err(format_error(&error)) + } + Err(_) => Ok(format!( + "Agent {task_id} is still running after waiting {timeout_ms} ms." + )), + } + }) + }), + source: ToolSource::Native, + } +} + +#[must_use] +pub(crate) fn make_task_stop_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { + RegisteredTool { + definition: definition( + NativeTool::TaskStop, + "Stop a running background agent by task ID.", + serde_json::json!({ + "type": "object", + "properties": { + "task_id": { + "type": "string", + "description": "The background agent task ID to stop." + } + }, + "required": ["task_id"], + "additionalProperties": false + }), + ), + executor: Arc::new(move |args, _ctx| { + let supervisor = supervisor.clone(); + Box::pin(async move { + let task_id = tools::required_str(&args, "task_id")?; + supervisor + .close_agent(task_id) + .await + .map_err(|error| format_error(&error))?; + Ok(format!("Agent {task_id} stopped.")) + }) + }), + source: ToolSource::Native, + } +} + +#[must_use] +pub(crate) fn make_send_message_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { + RegisteredTool { + definition: definition( + NativeTool::SendMessage, + "Send additional instructions to a running child agent by its Fabro agent ID.", + serde_json::json!({ + "type": "object", + "properties": { + "to": { + "type": "string", + "description": "The running Fabro agent ID." + }, + "message": { + "type": "string", + "description": "The follow-up message." + }, + "summary": { + "type": "string", + "maxLength": 200, + "description": "Optional short preview of the message." + } + }, + "required": ["to", "message"], + "additionalProperties": false + }), + ), + executor: Arc::new(move |args, _ctx| { + let supervisor = supervisor.clone(); + Box::pin(async move { + let recipient = tools::required_str(&args, "to")?; + let message = tools::required_str(&args, "message")?; + supervisor + .send_input(recipient, message) + .map_err(|error| format_error(&error))?; + Ok(format!("Message sent to agent {recipient}.")) + }) + }), + source: ToolSource::Native, + } +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeSet; + use std::sync::Mutex; + + use serde_json::json; + use tokio_util::sync::CancellationToken; + + use super::*; + use crate::sandbox::Sandbox; + use crate::test_support::{MockSandbox, make_session, text_response}; + use crate::todo_runtime::TodoRuntime; + use crate::todo_tools::{ + make_task_create_tool, make_task_get_tool, make_task_list_tool, make_task_update_tool, + }; + + fn property_names(tool: &RegisteredTool) -> BTreeSet<&str> { + tool.definition.parameters["properties"] + .as_object() + .unwrap() + .keys() + .map(String::as_str) + .collect() + } + + fn required_names(tool: &RegisteredTool) -> BTreeSet<&str> { + tool.definition.parameters["required"] + .as_array() + .map(|required| { + required + .iter() + .map(|value| value.as_str().unwrap()) + .collect() + }) + .unwrap_or_default() + } + + fn assert_schema(tool: &RegisteredTool, properties: &[&str], required: &[&str]) { + assert_eq!(tool.definition.parameters["type"], "object"); + assert_eq!( + tool.definition.parameters["additionalProperties"], + Value::Bool(false) + ); + assert_eq!(property_names(tool), properties.iter().copied().collect()); + assert_eq!(required_names(tool), required.iter().copied().collect()); + } + + fn context() -> ToolContext { + ToolContext { + env: Arc::new(MockSandbox::default()) as Arc, + cancel: CancellationToken::new(), + tool_env_provider: None, + session_id: Some("root".to_string()), + root_session_id: Some("root".to_string()), + tool_call_id: Some("call".to_string()), + agent_event_emitter: None, + } + } + + #[test] + fn core_adapter_schemas_match_the_claude5_contract() { + let options = NativeToolOptions::for_profile(fabro_model::AgentProfileKind::Claude5); + assert_schema(&make_read_tool(), &["file_path", "limit", "offset"], &[ + "file_path", + ]); + assert_schema(&make_write_tool(), &["content", "file_path"], &[ + "content", + "file_path", + ]); + assert_schema( + &make_edit_tool(), + &["file_path", "new_string", "old_string", "replace_all"], + &["file_path", "new_string", "old_string"], + ); + let bash = make_bash_tool(&options); + assert_schema(&bash, &["command", "description", "timeout"], &["command"]); + assert_eq!( + bash.definition.parameters["properties"]["timeout"]["maximum"], + 600_000 + ); + assert_schema(&make_web_fetch_tool(None), &["prompt", "url"], &[ + "prompt", "url", + ]); + assert_schema(&make_web_search_tool("key".to_string()), &["query"], &[ + "query", + ]); + + let todo_runtime = Arc::new(TodoRuntime::new()); + assert_schema( + &strict_object_tool(make_task_create_tool(todo_runtime.clone())), + &["activeForm", "description", "metadata", "subject"], + &["description", "subject"], + ); + assert_schema( + &strict_object_tool(make_task_update_tool(todo_runtime.clone())), + &[ + "activeForm", + "addBlockedBy", + "addBlocks", + "description", + "metadata", + "owner", + "status", + "subject", + "taskId", + ], + &["taskId"], + ); + assert_schema( + &strict_object_tool(make_task_get_tool(todo_runtime.clone())), + &["taskId"], + &["taskId"], + ); + assert_schema( + &strict_object_tool(make_task_list_tool(todo_runtime)), + &[], + &[], + ); + } + + #[test] + fn lifecycle_adapter_schemas_match_the_claude5_contract() { + let supervisor = SubAgentSupervisor::new(3); + let factory: SessionFactory = Arc::new(|| panic!("unused")); + assert_schema( + &make_agent_tool(supervisor.clone(), factory, 0), + &["description", "prompt", "run_in_background"], + &["description", "prompt"], + ); + assert_schema( + &make_task_output_tool(supervisor.clone()), + &["block", "task_id", "timeout"], + &["block", "task_id", "timeout"], + ); + assert_schema(&make_task_stop_tool(supervisor.clone()), &["task_id"], &[ + "task_id", + ]); + assert_schema( + &make_send_message_tool(supervisor), + &["message", "summary", "to"], + &["message", "to"], + ); + } + + #[tokio::test] + async fn agent_defaults_to_background_and_produces_parent_notification() { + let supervisor = SubAgentSupervisor::new(3); + let session = make_session(vec![text_response("child report")]).await; + let session_slot = Arc::new(Mutex::new(Some(session))); + let factory_slot = Arc::clone(&session_slot); + let factory: SessionFactory = Arc::new(move || { + factory_slot + .lock() + .unwrap() + .take() + .expect("factory should be called once") + }); + let tool = make_agent_tool(supervisor.clone(), factory, 0); + + let output = (tool.executor)( + json!({ + "description": "Inspect child", + "prompt": "Inspect the child task" + }), + context(), + ) + .await + .unwrap(); + + let task_id = output + .strip_prefix("Agent started in the background.\n\nTask ID: ") + .expect("Agent should return a background task ID"); + let notifications = supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .unwrap(); + assert_eq!(notifications.len(), 1); + assert_eq!(notifications[0].agent_id, task_id); + assert_eq!(notifications[0].description, "Inspect child"); + assert_eq!( + notifications[0].result.as_ref().unwrap().output, + "child report" + ); + + supervisor.shutdown_all().await; + } + + #[tokio::test] + async fn task_output_suppresses_a_racing_automatic_notification() { + let supervisor = SubAgentSupervisor::new(3); + let session = make_session(vec![text_response("explicit report")]).await; + let task_id = supervisor + .spawn_with_parent_notification( + session, + "Inspect".to_string(), + "Inspect explicitly".to_string(), + 0, + ) + .unwrap(); + supervisor + .wait_with_cancel(&task_id, &CancellationToken::new()) + .await + .unwrap(); + + let tool = make_task_output_tool(supervisor.clone()); + let output = (tool.executor)( + json!({ + "task_id": task_id, + "block": false, + "timeout": 0 + }), + context(), + ) + .await + .unwrap(); + + assert!(output.contains("explicit report")); + assert!( + supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .is_none() + ); + + supervisor.shutdown_all().await; + } +} diff --git a/lib/components/fabro-agent/src/profiles/mod.rs b/lib/components/fabro-agent/src/profiles/mod.rs index 9defbf54f..f3ff26aa1 100644 --- a/lib/components/fabro-agent/src/profiles/mod.rs +++ b/lib/components/fabro-agent/src/profiles/mod.rs @@ -4,6 +4,8 @@ use std::sync::Arc; use fabro_model::{AgentProfileKind, Catalog, CodecKind, ProviderId}; pub mod anthropic; +pub mod claude5; +pub(crate) mod claude5_tools; pub mod gemini; pub mod gpt56; pub mod kimi; @@ -11,6 +13,7 @@ pub mod kimi_tools; pub mod openai; pub use anthropic::AnthropicProfile; +pub use claude5::Claude5Profile; pub use gemini::GeminiProfile; pub use gpt56::Gpt56Profile; pub use kimi::KimiProfile; @@ -22,6 +25,7 @@ use crate::config::{NativeToolOptions, ToolSecrets}; use crate::native_tool::{NativeTool, ToolVocabulary}; use crate::sandbox::Sandbox; use crate::skills::{Skill, format_skills_prompt_section}; +use crate::todo_runtime::TodoRuntime; use crate::tool_registry::ToolRegistry; use crate::tools::{self, WebFetchSummarizer}; @@ -39,6 +43,7 @@ pub struct AgentProfileBuilder { catalog: Arc, native_tool_options: NativeToolOptions, summarizer: Option, + todo_runtime: Arc, } impl AgentProfileBuilder { @@ -56,6 +61,7 @@ impl AgentProfileBuilder { catalog, native_tool_options: NativeToolOptions::for_profile(profile_kind), summarizer: None, + todo_runtime: Arc::new(TodoRuntime::new()), } } @@ -99,6 +105,16 @@ impl AgentProfileBuilder { .with_provider_id(self.provider_id.clone()) .with_catalog(Arc::clone(&self.catalog)), ), + AgentProfileKind::Claude5 => Box::new( + Claude5Profile::with_native_tools_and_todo_runtime( + model, + options, + summarizer, + Arc::clone(&self.todo_runtime), + ) + .with_provider_id(self.provider_id.clone()) + .with_catalog(Arc::clone(&self.catalog)), + ), AgentProfileKind::Kimi => Box::new( KimiProfile::with_native_tools(model, options, summarizer) .with_provider_id(self.provider_id.clone()) @@ -374,11 +390,13 @@ pub fn build_env_context_block_with(env: &dyn Sandbox, ctx: &EnvContext) -> Stri mod tests { use fabro_llm::types::ToolDefinition; use fabro_model::catalog::LlmCatalogSettings; + use tokio_util::sync::CancellationToken; use super::*; + use crate::question_tools; use crate::subagent::{SessionFactory, SubAgentSupervisor}; use crate::test_support::MockSandbox; - use crate::tools::WEB_SEARCH_TOOL_NAME; + use crate::tool_registry::ToolContext; fn native_tool_options( profile_kind: AgentProfileKind, @@ -411,6 +429,25 @@ mod tests { profile } + fn claude5_profile( + has_web_search: bool, + has_subagents: bool, + has_question: bool, + ) -> Claude5Profile { + let options = native_tool_options(AgentProfileKind::Claude5, has_web_search); + let mut profile = Claude5Profile::with_native_tools("claude-sonnet-5", &options, None); + if has_subagents { + register_test_subagent_tools(&mut profile); + } + if has_question { + question_tools::register_question_tools( + AgentProfileKind::Claude5, + profile.tool_registry_mut(), + ); + } + profile + } + fn gemini_profile(has_web_search: bool) -> GeminiProfile { let options = native_tool_options(AgentProfileKind::Gemini, has_web_search); GeminiProfile::with_native_tools("gemini-3-flash-preview", &options, None) @@ -595,6 +632,11 @@ mod tests { ProviderId::gemini(), "gemini-3-flash-preview", ), + ( + AgentProfileKind::Claude5, + ProviderId::anthropic(), + "claude-sonnet-5", + ), (AgentProfileKind::Gpt56, ProviderId::openai(), "gpt-5.6-sol"), ]; @@ -606,12 +648,13 @@ mod tests { Arc::clone(&catalog), ) .build(); + let web_search_name = NativeTool::WebSearch.name(profile.tool_registry().vocabulary()); assert_eq!(profile.profile_kind(), profile_kind); assert_eq!(profile.provider_id(), provider_id); - assert!(profile.tool_registry().get(WEB_SEARCH_TOOL_NAME).is_none()); + assert!(profile.tool_registry().get(web_search_name).is_none()); let prompt = profile.build_system_prompt(&env, &EnvContext::default(), &[], None, &[]); assert!( - !prompt.contains("web_search"), + !prompt.contains(web_search_name), "{profile_kind:?} prompt advertised an unavailable tool" ); @@ -627,22 +670,79 @@ mod tests { // Built twice: one configured builder must outfit both a root // session and the child sessions it spawns. for configured in [configured_builder.build(), configured_builder.build()] { - assert!( - configured - .tool_registry() - .get(WEB_SEARCH_TOOL_NAME) - .is_some() - ); + assert!(configured.tool_registry().get(web_search_name).is_some()); let prompt = configured.build_system_prompt(&env, &EnvContext::default(), &[], None, &[]); assert!( - prompt.contains("web_search"), + prompt.contains(web_search_name), "{profile_kind:?} prompt omitted guidance for an available tool" ); } } } + #[tokio::test] + async fn claude5_builder_shares_tasks_across_root_and_child_profiles() { + let builder = AgentProfileBuilder::new( + AgentProfileKind::Claude5, + ProviderId::anthropic(), + "claude-sonnet-5", + Arc::new(Catalog::from_builtin().unwrap()), + ); + let root = builder.build(); + let child = builder.build(); + let root_create = Arc::clone( + &root + .tool_registry() + .get("TaskCreate") + .expect("root should expose TaskCreate") + .executor, + ); + let child_create = Arc::clone( + &child + .tool_registry() + .get("TaskCreate") + .expect("child should expose TaskCreate") + .executor, + ); + let child_list = Arc::clone( + &child + .tool_registry() + .get("TaskList") + .expect("child should expose TaskList") + .executor, + ); + let env: Arc = Arc::new(MockSandbox::default()); + let context = |session_id: &str| ToolContext { + env: Arc::clone(&env), + cancel: CancellationToken::new(), + tool_env_provider: None, + session_id: Some(session_id.to_string()), + root_session_id: Some("root-session".to_string()), + tool_call_id: None, + agent_event_emitter: None, + }; + + root_create( + serde_json::json!({"subject": "Parent task", "description": "Root work"}), + context("root-session"), + ) + .await + .unwrap(); + child_create( + serde_json::json!({"subject": "Child task", "description": "Child work"}), + context("child-session"), + ) + .await + .unwrap(); + let tasks = child_list(serde_json::json!({}), context("child-session")) + .await + .unwrap(); + + assert!(tasks.contains("#1 [pending] Parent task"), "{tasks}"); + assert!(tasks.contains("#2 [pending] Child task"), "{tasks}"); + } + #[test] fn profile_builder_selects_a_codec_compatible_gpt56_editor() { let overrides: LlmCatalogSettings = @@ -680,6 +780,46 @@ mod tests { insta::assert_snapshot!(system_prompt(&anthropic_profile(true, true))); } + #[test] + fn claude5_default_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(false, false, false))); + } + + #[test] + fn claude5_web_search_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(true, false, false))); + } + + #[test] + fn claude5_subagents_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(false, true, false))); + } + + #[test] + fn claude5_question_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(false, false, true))); + } + + #[test] + fn claude5_web_search_and_subagents_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(true, true, false))); + } + + #[test] + fn claude5_web_search_and_question_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(true, false, true))); + } + + #[test] + fn claude5_subagents_and_question_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(false, true, true))); + } + + #[test] + fn claude5_all_conditionals_prompt_snapshot() { + insta::assert_snapshot!(system_prompt(&claude5_profile(true, true, true))); + } + #[test] fn gemini_default_prompt_snapshot() { insta::assert_snapshot!(system_prompt(&gemini_profile(false))); diff --git a/lib/components/fabro-agent/src/profiles/prompts/claude5.md.j2 b/lib/components/fabro-agent/src/profiles/prompts/claude5.md.j2 new file mode 100644 index 000000000..143be4f55 --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/prompts/claude5.md.j2 @@ -0,0 +1,80 @@ +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + +{{ inputs.env_block }} + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. +{% if inputs.has_web_search %} +Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. +{% endif %} + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + +{% if inputs.has_agent %} +# Background agents + +Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. + +Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. + +Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. + +An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. +{% endif %} + +{% if inputs.has_ask_user_question %} +# Asking the user + +Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. + +When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. +{% endif %} + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_all_conditionals_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_all_conditionals_prompt_snapshot.snap new file mode 100644 index 000000000..8d4a27c3f --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_all_conditionals_prompt_snapshot.snap @@ -0,0 +1,89 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(true, true, true))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + +Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + +# Background agents + +Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. + +Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. + +Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. + +An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. + + + +# Asking the user + +Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. + +When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_default_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_default_prompt_snapshot.snap new file mode 100644 index 000000000..2152f374d --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_default_prompt_snapshot.snap @@ -0,0 +1,71 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(false, false, false))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + + + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap new file mode 100644 index 000000000..c1628ad2a --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap @@ -0,0 +1,77 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(false, false, true))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + + + +# Asking the user + +Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. + +When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap new file mode 100644 index 000000000..95d827e7a --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap @@ -0,0 +1,87 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(false, true, true))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + +# Background agents + +Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. + +Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. + +Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. + +An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. + + + +# Asking the user + +Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. + +When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap new file mode 100644 index 000000000..eaaaf1cbb --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap @@ -0,0 +1,81 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(false, true, false))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + +# Background agents + +Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. + +Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. + +Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. + +An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. + + + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap new file mode 100644 index 000000000..041ac14ff --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap @@ -0,0 +1,79 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(true, false, true))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + +Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + + + +# Asking the user + +Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. + +When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap new file mode 100644 index 000000000..d67ec831e --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap @@ -0,0 +1,83 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(true, true, false))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + +Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + +# Background agents + +Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. + +Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. + +Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. + +An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. + + + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap new file mode 100644 index 000000000..b92b72e73 --- /dev/null +++ b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap @@ -0,0 +1,73 @@ +--- +source: lib/components/fabro-agent/src/profiles/mod.rs +expression: "system_prompt(&claude5_profile(true, false, false))" +--- +You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. + +When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. + + +Working directory: /home/test +Is git repository: false +Platform: linux +OS version: Linux 6.1.0 + + +# Harness + +- Text outside tool calls is shown to the user as GitHub-flavored Markdown. +- The user may not see your reasoning or raw tool output. Make the final response self-contained. +- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. +- Follow all project and user instructions included in this prompt. +- Reference code with `file_path:line_number` when a precise location helps. + +# Delivering work + +Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. + +Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. + +Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. + +# Working in the codebase + +- Read relevant code before proposing or making changes. +- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. +- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. +- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. +- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. +- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. +- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. + +# Tool use + +Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. + +Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. + +Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. + +Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. + +Use `WebFetch` with both a URL and a prompt describing the information to extract. + +Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. + + +Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. + + + + + +# Communicating with the user + +Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. + +Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. + +Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. + +# Context management + +Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/question_tools.rs b/lib/components/fabro-agent/src/question_tools.rs index d4ff2f274..3fcc2980f 100644 --- a/lib/components/fabro-agent/src/question_tools.rs +++ b/lib/components/fabro-agent/src/question_tools.rs @@ -149,6 +149,28 @@ struct AnthropicOption { preview: Option, } +#[derive(Debug, Deserialize)] +struct Claude5QuestionToolArgs { + questions: Vec, +} + +#[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +struct Claude5Question { + question: String, + header: String, + options: Vec, + multi_select: bool, +} + +#[derive(Debug, Deserialize)] +struct Claude5Option { + label: String, + description: String, + #[serde(default)] + preview: Option, +} + #[must_use] pub fn is_question_tool(name: &str) -> bool { matches!( @@ -168,6 +190,9 @@ pub fn register_question_tools(profile_kind: AgentProfileKind, registry: &mut To AgentProfileKind::Anthropic | AgentProfileKind::Kimi => { registry.register(make_anthropic_question_tool()); } + AgentProfileKind::Claude5 => { + registry.register(make_claude5_question_tool()); + } AgentProfileKind::Gemini => {} } } @@ -269,6 +294,82 @@ fn make_anthropic_question_tool() -> RegisteredTool { } } +fn make_claude5_question_tool() -> RegisteredTool { + RegisteredTool { + definition: ToolDefinition { + name: ANTHROPIC_ASK_USER_QUESTION_TOOL.to_string(), + description: "Ask the human up to four questions when a decision is genuinely theirs to make. The UI automatically provides an Other option for custom text.".to_string(), + parameters: json!({ + "type": "object", + "properties": { + "questions": { + "description": "Questions to ask the user (1-4 questions)", + "type": "array", + "minItems": 1, + "maxItems": 4, + "items": { + "type": "object", + "properties": { + "question": { + "description": "The complete, clear, and specific question to ask.", + "type": "string" + }, + "header": { + "description": "Very short label displayed as a chip/tag (max 12 chars).", + "type": "string" + }, + "options": { + "description": "Two to four choices. Do not include Other; the UI adds it automatically.", + "type": "array", + "minItems": 2, + "maxItems": 4, + "items": { + "type": "object", + "properties": { + "label": { + "description": "Concise display text for the option.", + "type": "string" + }, + "description": { + "description": "What the option means and its relevant trade-offs.", + "type": "string" + }, + "preview": { + "description": "Optional Markdown preview for single-select visual comparisons.", + "type": "string" + } + }, + "required": ["label", "description"], + "additionalProperties": false + } + }, + "multiSelect": { + "description": "Whether the user may select multiple options.", + "default": false, + "type": "boolean" + } + }, + "required": ["question", "header", "options", "multiSelect"], + "additionalProperties": false + } + } + }, + "required": ["questions"], + "additionalProperties": false + }), + }, + executor: Arc::new(|args, ctx| { + Box::pin(async move { + let parsed: Claude5QuestionToolArgs = parse_tool_args(args)?; + let questions = normalize_claude5_questions(parsed)?; + let answers = execute_question_tool(ctx, questions).await?; + format_anthropic_answers(&answers) + }) + }), + source: ToolSource::Native, + } +} + fn parse_tool_args Deserialize<'de>>(args: serde_json::Value) -> Result { serde_json::from_value(args).map_err(|err| format!("invalid question tool arguments: {err}")) } @@ -350,6 +451,71 @@ fn normalize_anthropic_questions( .collect() } +fn normalize_claude5_questions( + args: Claude5QuestionToolArgs, +) -> Result, String> { + if !(1..=4).contains(&args.questions.len()) { + return Err("questions must contain between one and four questions".to_string()); + } + + args.questions + .into_iter() + .map(|question| { + let original_question = non_empty(&question.question, "question")?; + let header = non_empty(&question.header, "question header")?; + if header.chars().count() > 12 { + return Err("question header must contain at most 12 characters".to_string()); + } + if !(2..=4).contains(&question.options.len()) { + return Err("each question must contain between two and four options".to_string()); + } + if question.multi_select + && question + .options + .iter() + .any(|option| option.preview.is_some()) + { + return Err( + "option previews are not supported for multi-select questions".to_string(), + ); + } + + let options = question + .options + .into_iter() + .enumerate() + .map(|(idx, option)| { + Ok(InterviewOption { + key: option_key(idx), + label: non_empty(&option.label, "option label")?, + description: Some(bounded_display_field( + &non_empty(&option.description, "option description")?, + OPTION_DESCRIPTION_MAX_CHARS, + )), + preview: option + .preview + .map(|value| bounded_display_field(&value, OPTION_PREVIEW_MAX_CHARS)), + }) + }) + .collect::, String>>()?; + + Ok(AgentQuestion { + original_id: None, + text: display_text(Some(&header), &original_question), + header: Some(header), + original_question, + question_type: if question.multi_select { + QuestionType::MultiSelect + } else { + QuestionType::MultipleChoice + }, + options, + allow_freeform: true, + }) + }) + .collect() +} + fn options_from_openai(options: Vec) -> Vec { options .into_iter() @@ -472,6 +638,7 @@ fn format_anthropic_answers(answers: &[AgentQuestionAnswer]) -> Result, @@ -601,8 +768,121 @@ mod tests { assert!(kimi.get(ANTHROPIC_ASK_USER_QUESTION_TOOL).is_some()); assert!(kimi.get(OPENAI_REQUEST_USER_INPUT_TOOL).is_none()); + let mut claude5 = ToolRegistry::with_vocabulary(ToolVocabulary::Claude5); + register_question_tools(AgentProfileKind::Claude5, &mut claude5); + let tool = claude5.get(ANTHROPIC_ASK_USER_QUESTION_TOOL).unwrap(); + assert_eq!(tool.definition.parameters["additionalProperties"], false); + assert_eq!( + tool.definition.parameters["properties"] + .as_object() + .unwrap() + .keys() + .map(String::as_str) + .collect::>(), + vec!["questions"] + ); + assert_eq!( + tool.definition.parameters["properties"]["questions"]["maxItems"], + 4 + ); + assert!(claude5.get(OPENAI_REQUEST_USER_INPUT_TOOL).is_none()); + let mut gemini = ToolRegistry::new(); register_question_tools(AgentProfileKind::Gemini, &mut gemini); assert!(gemini.names().is_empty()); } + + #[test] + fn claude5_question_contract_is_strict_and_preserves_preview() { + let args: Claude5QuestionToolArgs = serde_json::from_value(json!({ + "questions": [{ + "header": "Approach", + "question": "Which approach should we use?", + "multiSelect": false, + "options": [ + { + "label": "Simple", + "description": "Use the smallest implementation.", + "preview": "fn simple() {}" + }, + { + "label": "Flexible", + "description": "Allow future extension." + } + ] + }] + })) + .unwrap(); + + let questions = normalize_claude5_questions(args).unwrap(); + + assert_eq!(questions[0].header.as_deref(), Some("Approach")); + assert_eq!( + questions[0].options[0].preview.as_deref(), + Some("fn simple() {}") + ); + assert!(questions[0].allow_freeform); + } + + #[test] + fn claude5_rejects_previews_for_multi_select_questions() { + let args: Claude5QuestionToolArgs = serde_json::from_value(json!({ + "questions": [{ + "header": "Features", + "question": "Which features should we enable?", + "multiSelect": true, + "options": [ + { + "label": "Auth", + "description": "Enable authentication.", + "preview": "auth = true" + }, + { + "label": "Metrics", + "description": "Enable metrics." + } + ] + }] + })) + .unwrap(); + + assert!(normalize_claude5_questions(args).is_err()); + } + + #[tokio::test] + async fn claude5_question_tool_rejects_subagent_sessions() { + let tool = make_claude5_question_tool(); + let error = (tool.executor)( + json!({ + "questions": [{ + "header": "Approach", + "question": "Which approach?", + "multiSelect": false, + "options": [ + { + "label": "Simple", + "description": "Use the simple approach." + }, + { + "label": "Flexible", + "description": "Use the flexible approach." + } + ] + }] + }), + ToolContext { + env: Arc::new(MockSandbox::default()), + cancel: CancellationToken::new(), + tool_env_provider: None, + session_id: Some("child".to_string()), + root_session_id: Some("root".to_string()), + tool_call_id: Some("call".to_string()), + agent_event_emitter: None, + }, + ) + .await + .unwrap_err(); + + assert!(error.contains("only available to the root agent")); + } } diff --git a/lib/components/fabro-agent/src/session.rs b/lib/components/fabro-agent/src/session.rs index 4914b5638..f04e99257 100644 --- a/lib/components/fabro-agent/src/session.rs +++ b/lib/components/fabro-agent/src/session.rs @@ -47,7 +47,10 @@ use crate::skills::{ ExpandedInput, Skill, default_skill_dirs, discover_skills, expand_skill, make_use_skill_tool_for_vocabulary, }; -use crate::subagent::{SubAgentCallbackEvent, SubAgentEventCallback, SubAgentSupervisor}; +use crate::subagent::{ + SubAgentCallbackEvent, SubAgentEventCallback, SubAgentSupervisor, + format_parent_notification_batch, +}; use crate::tool_execution::execute_tool_calls; use crate::tool_permissions::canonical_tool_name; use crate::tool_registry::ToolDefinitionWithSource; @@ -1318,7 +1321,10 @@ impl Session { }) }); - // Process the initial input, then drain any followups + // Process the initial input, then drain followups. Claude-compatible + // background-agent results join this same boundary queue: they never + // interrupt inference or a tool call, and all results already ready at + // a boundary are delivered in one additional parent turn. let mut result = self .run_single_input( input, @@ -1336,10 +1342,33 @@ impl Session { .lock() .expect("followup queue lock poisoned") .pop_front(); - let Some(followup) = followup else { break }; + let next_input = if let Some(followup) = followup { + Some(followup) + } else if let Some(supervisor) = self.subagent_supervisor.clone() { + match supervisor + .next_parent_notification_batch(&self.cancel_token) + .await + { + Ok(Some(notifications)) => { + Some(format_parent_notification_batch(¬ifications)) + } + Ok(None) => None, + Err(Error::Interrupted(InterruptReason::Cancelled)) => { + result = Err(self.interrupted_error()); + None + } + Err(error) => { + result = Err(error); + None + } + } + } else { + None + }; + let Some(next_input) = next_input else { break }; result = self .run_single_input( - &followup, + &next_input, &agent_tool_runtime, &mut timing, &mut usage, @@ -3040,6 +3069,70 @@ mod tests { ); } + #[tokio::test] + async fn background_agent_notifications_are_batched_into_one_parent_turn() { + let supervisor = SubAgentSupervisor::new(3); + let first = make_session(vec![text_response("first result")]).await; + let second = make_session(vec![text_response("second result")]).await; + let first_id = supervisor + .spawn_with_parent_notification( + first, + "first task".to_string(), + "Inspect first".to_string(), + 0, + ) + .unwrap(); + let second_id = supervisor + .spawn_with_parent_notification( + second, + "second task".to_string(), + "Inspect second".to_string(), + 0, + ) + .unwrap(); + + // Make both results ready before the parent reaches its safe turn + // boundary so batching is deterministic. + supervisor + .wait_with_cancel(&first_id, &CancellationToken::new()) + .await + .unwrap(); + supervisor + .wait_with_cancel(&second_id, &CancellationToken::new()) + .await + .unwrap(); + + let provider = Arc::new(ScriptedStreamProvider::new(vec![ + ScriptedStreamCall::Response(Box::new(text_response("Parent is waiting"))), + ScriptedStreamCall::Response(Box::new(text_response("Synthesized both results"))), + ])); + let mut parent = + make_session_with_provider_and_manager(provider, Some(supervisor.clone())).await; + + let output = parent + .process_input_with_output("Delegate both tasks") + .await + .unwrap(); + + assert_eq!(output.as_deref(), Some("Synthesized both results")); + let turns = parent.history().turns(); + assert_eq!(turns.len(), 4); + let Message::User { + content: notification, + .. + } = &turns[2] + else { + panic!("third turn should deliver the background results"); + }; + assert_eq!(notification.matches("").count(), 2); + assert!(notification.contains(&first_id)); + assert!(notification.contains(&second_id)); + assert!(notification.contains("first result")); + assert!(notification.contains("second result")); + + supervisor.shutdown_all().await; + } + #[tokio::test] async fn events_emitted() { let mut session = make_session(vec![text_response("Hello")]).await; diff --git a/lib/components/fabro-agent/src/skills.rs b/lib/components/fabro-agent/src/skills.rs index ecd08e9cb..f1aff7031 100644 --- a/lib/components/fabro-agent/src/skills.rs +++ b/lib/components/fabro-agent/src/skills.rs @@ -189,6 +189,24 @@ pub fn make_use_skill_tool_for_vocabulary( "required": ["skill_name"] }), ), + ToolVocabulary::Claude5 => ( + "skill", + serde_json::json!({ + "type": "object", + "properties": { + "skill": { + "type": "string", + "description": "Exact name of the skill to invoke" + }, + "args": { + "type": "string", + "description": "Optional argument string to pass to the skill" + } + }, + "required": ["skill"], + "additionalProperties": false + }), + ), ToolVocabulary::KimiCode => ( "skill", serde_json::json!({ @@ -730,4 +748,46 @@ name: trimmed .is_none() ); } + + #[tokio::test] + async fn claude5_skill_schema_uses_skill_and_optional_args() { + let skills = Arc::new(test_skills()); + let tool = make_use_skill_tool_for_vocabulary(skills, ToolVocabulary::Claude5); + let result = (tool.executor)( + serde_json::json!({"skill": "commit", "args": "only staged files"}), + ToolContext { + env: Arc::new(MockSandbox::default()), + cancel: CancellationToken::new(), + tool_env_provider: None, + session_id: None, + root_session_id: None, + tool_call_id: None, + agent_event_emitter: None, + }, + ) + .await + .unwrap(); + + assert!(result.contains("only staged files"), "{result}"); + assert_eq!( + tool.definition.parameters["required"], + serde_json::json!(["skill"]) + ); + assert_eq!(tool.definition.parameters["additionalProperties"], false); + assert!( + tool.definition.parameters["properties"] + .get("skill") + .is_some() + ); + assert!( + tool.definition.parameters["properties"] + .get("args") + .is_some() + ); + assert!( + tool.definition.parameters["properties"] + .get("skill_name") + .is_none() + ); + } } diff --git a/lib/components/fabro-agent/src/subagent.rs b/lib/components/fabro-agent/src/subagent.rs index 3cb2af0da..0d91d58e8 100644 --- a/lib/components/fabro-agent/src/subagent.rs +++ b/lib/components/fabro-agent/src/subagent.rs @@ -3,6 +3,7 @@ use std::sync::{Arc, Mutex, RwLock}; use std::time::Duration; use fabro_llm::types::ToolDefinition; +use fabro_util::error as util_error; use futures::future; use tokio::sync::{oneshot, watch}; use tokio::task::{AbortHandle, JoinHandle}; @@ -32,6 +33,53 @@ pub struct SubAgentResult { pub turns_used: usize, } +/// A terminal background-agent result waiting to be delivered to its parent at +/// a safe turn boundary. +#[derive(Debug, Clone)] +pub(crate) struct SubAgentParentNotification { + pub agent_id: String, + pub description: String, + pub result: Result, +} + +pub(crate) fn format_parent_notification_batch( + notifications: &[SubAgentParentNotification], +) -> String { + notifications + .iter() + .map(|notification| { + let (status, result) = match ¬ification.result { + Ok(result) if result.success => ("completed", result.output.clone()), + Ok(result) => ("failed", result.output.clone()), + Err(error) => ("failed", util_error::collect_chain(error).join(": ")), + }; + format!( + "\n {}\n {status}\n \ + {}\n {}\n", + escape_notification_xml(¬ification.agent_id), + escape_notification_xml(¬ification.description), + escape_notification_xml(&result), + ) + }) + .collect::>() + .join("\n\n") +} + +fn escape_notification_xml(value: &str) -> String { + let mut escaped = String::with_capacity(value.len()); + for character in value.chars() { + match character { + '&' => escaped.push_str("&"), + '<' => escaped.push_str("<"), + '>' => escaped.push_str(">"), + '"' => escaped.push_str("""), + '\'' => escaped.push_str("'"), + _ => escaped.push(character), + } + } + escaped +} + #[derive(Debug, Clone)] pub enum SubAgentStatus { Running, @@ -76,6 +124,113 @@ struct SupervisorState { agents: HashMap, } +#[derive(Default)] +struct ParentNotificationState { + pending: HashMap, + ready: VecDeque, +} + +struct ParentNotificationHub { + state: Mutex, + changed: watch::Sender, +} + +impl ParentNotificationHub { + fn new() -> Self { + let (changed, _) = watch::channel(0); + Self { + state: Mutex::new(ParentNotificationState::default()), + changed, + } + } + + fn register(&self, agent_id: String, description: String) { + self.state + .lock() + .expect("parent notification lock poisoned") + .pending + .insert(agent_id, description); + self.signal(); + } + + fn complete(&self, agent_id: &str, result: Result) { + { + let mut state = self + .state + .lock() + .expect("parent notification lock poisoned"); + let Some(description) = state.pending.remove(agent_id) else { + return; + }; + state.ready.push_back(SubAgentParentNotification { + agent_id: agent_id.to_string(), + description, + result, + }); + } + self.signal(); + } + + fn suppress(&self, agent_id: &str) { + let changed = { + let mut state = self + .state + .lock() + .expect("parent notification lock poisoned"); + let removed_pending = state.pending.remove(agent_id).is_some(); + let ready_len = state.ready.len(); + state + .ready + .retain(|notification| notification.agent_id != agent_id); + removed_pending || state.ready.len() != ready_len + }; + if changed { + self.signal(); + } + } + + async fn next_batch( + &self, + cancel: &CancellationToken, + ) -> Result>, Error> { + let mut changed = self.changed.subscribe(); + loop { + { + let mut state = self + .state + .lock() + .expect("parent notification lock poisoned"); + if !state.ready.is_empty() { + return Ok(Some(state.ready.drain(..).collect())); + } + if state.pending.is_empty() { + return Ok(None); + } + } + + tokio::select! { + biased; + () = cancel.cancelled() => { + return Err(Error::Interrupted(InterruptReason::Cancelled)); + } + observed = changed.changed() => { + observed.map_err(|_| { + Error::InvalidState( + "Background-agent notification observer closed unexpectedly".to_string(), + ) + })?; + } + } + } + } + + fn signal(&self) { + self.changed.send_modify(|generation| { + *generation = generation.wrapping_add(1); + }); + } +} + struct ShutdownWork { agent_id: String, depth: usize, @@ -119,6 +274,7 @@ fn spawn_result_monitor( child_task: JoinHandle>, status: watch::Sender, event_callback: Arc>>, + parent_notifications: Arc, agent_id: String, depth: usize, ) -> JoinHandle<()> { @@ -141,17 +297,17 @@ fn spawn_result_monitor( return; } - let event = match task_result { + let event = match &task_result { Ok(result) => AgentEvent::SubAgentCompleted { - agent_id, + agent_id: agent_id.clone(), depth, success: result.success, turns_used: result.turns_used, }, Err(error) => AgentEvent::SubAgentFailed { - agent_id, + agent_id: agent_id.clone(), depth, - error, + error: error.clone(), }, }; let callback = event_callback @@ -161,6 +317,7 @@ fn spawn_result_monitor( if let Some(callback) = callback { callback(SubAgentCallbackEvent::Lifecycle(event)); } + parent_notifications.complete(&agent_id, task_result); }) } @@ -171,9 +328,10 @@ fn spawn_result_monitor( /// happen after the guard has been released. #[derive(Clone)] pub struct SubAgentSupervisor { - state: Arc>, - max_depth: usize, - event_callback: Arc>>, + state: Arc>, + max_depth: usize, + event_callback: Arc>>, + parent_notifications: Arc, } impl SubAgentSupervisor { @@ -183,6 +341,7 @@ impl SubAgentSupervisor { state: Arc::new(Mutex::new(SupervisorState::default())), max_depth, event_callback: Arc::new(RwLock::new(None)), + parent_notifications: Arc::new(ParentNotificationHub::new()), } } @@ -205,10 +364,32 @@ impl SubAgentSupervisor { } pub fn spawn( + &self, + session: Session, + task_prompt: String, + depth: usize, + ) -> Result { + self.spawn_inner(session, task_prompt, depth, None) + } + + /// Spawn a child whose terminal result should automatically be delivered + /// to the parent session. + pub(crate) fn spawn_with_parent_notification( + &self, + session: Session, + task_prompt: String, + description: String, + depth: usize, + ) -> Result { + self.spawn_inner(session, task_prompt, depth, Some(description)) + } + + fn spawn_inner( &self, mut session: Session, task_prompt: String, depth: usize, + parent_notification_description: Option, ) -> Result { if depth >= self.max_depth { return Err(Error::InvalidState(format!( @@ -295,6 +476,7 @@ impl SubAgentSupervisor { child_task, status.clone(), Arc::clone(&self.event_callback), + Arc::clone(&self.parent_notifications), agent_id.clone(), child_depth, ); @@ -314,6 +496,10 @@ impl SubAgentSupervisor { depth: child_depth, }); } + if let Some(description) = parent_notification_description { + self.parent_notifications + .register(agent_id.clone(), description); + } self.emit_event(AgentEvent::SubAgentSpawned { agent_id: agent_id.clone(), @@ -397,6 +583,22 @@ impl SubAgentSupervisor { } } + /// Stop automatic delivery for an agent whose result the parent explicitly + /// retrieved. Removes a result that may already have raced into the ready + /// queue. + pub(crate) fn suppress_parent_notification(&self, agent_id: &str) { + self.parent_notifications.suppress(agent_id); + } + + /// Wait until all currently-ready background results can be delivered in + /// one parent turn, or return `None` once no notifiable agents remain. + pub(crate) async fn next_parent_notification_batch( + &self, + cancel: &CancellationToken, + ) -> Result>, Error> { + self.parent_notifications.next_batch(cancel).await + } + #[cfg(test)] async fn wait(&self, agent_id: &str) -> Result { self.wait_with_cancel(agent_id, &CancellationToken::new()) @@ -404,6 +606,7 @@ impl SubAgentSupervisor { } fn begin_shutdown(&self, agent_id: &str, strict: bool) -> Result { + self.parent_notifications.suppress(agent_id); let mut state = self.state.lock().expect("subagent state lock poisoned"); let agent = state.agents.get_mut(agent_id).ok_or_else(|| { Error::InvalidState(format!( @@ -628,6 +831,7 @@ impl SubAgentSupervisor { child_task, status.clone(), Arc::clone(&self.event_callback), + Arc::clone(&self.parent_notifications), agent_id.clone(), depth, ); @@ -842,6 +1046,69 @@ mod tests { assert!(manager.is_empty()); } + #[tokio::test] + async fn parent_notifications_are_exactly_once_and_xml_escaped() { + let hub = ParentNotificationHub::new(); + hub.register("agent<&".to_string(), "Review & tests".to_string()); + let result = Ok(SubAgentResult { + output: "done & \"verified\"".to_string(), + success: true, + turns_used: 2, + }); + hub.complete("agent<&", result.clone()); + hub.complete("agent<&", result); + + let notifications = hub + .next_batch(&CancellationToken::new()) + .await + .unwrap() + .unwrap(); + assert_eq!(notifications.len(), 1); + let envelope = format_parent_notification_batch(¬ifications); + assert!(envelope.contains("completed")); + assert!(envelope.contains("agent<&")); + assert!(envelope.contains("Review <core> & tests")); + assert!( + envelope.contains("done <safely> & "verified"") + ); + assert!( + hub.next_batch(&CancellationToken::new()) + .await + .unwrap() + .is_none() + ); + } + + #[tokio::test] + async fn suppress_removes_pending_and_ready_parent_notifications() { + let hub = ParentNotificationHub::new(); + hub.register("pending".to_string(), "Pending".to_string()); + hub.suppress("pending"); + assert!( + hub.next_batch(&CancellationToken::new()) + .await + .unwrap() + .is_none() + ); + + hub.register("ready".to_string(), "Ready".to_string()); + hub.complete( + "ready", + Ok(SubAgentResult { + output: "done".to_string(), + success: true, + turns_used: 1, + }), + ); + hub.suppress("ready"); + assert!( + hub.next_batch(&CancellationToken::new()) + .await + .unwrap() + .is_none() + ); + } + #[tokio::test] async fn spawn_creates_agent_and_returns_id() { let manager = SubAgentSupervisor::new(3); diff --git a/lib/components/fabro-agent/src/todo_runtime.rs b/lib/components/fabro-agent/src/todo_runtime.rs index 760f2eb20..0d1d79e11 100644 --- a/lib/components/fabro-agent/src/todo_runtime.rs +++ b/lib/components/fabro-agent/src/todo_runtime.rs @@ -21,17 +21,33 @@ use crate::types::AgentEvent; /// `Arc` into each tool closure that needs it. #[derive(Debug, Default)] pub struct TodoRuntime { - lists: Mutex>, + lists: Mutex>, + task_counters: Mutex>, } impl TodoRuntime { #[must_use] pub fn new() -> Self { Self { - lists: Mutex::new(BTreeMap::new()), + lists: Mutex::new(BTreeMap::new()), + task_counters: Mutex::new(BTreeMap::new()), } } + /// Allocate the next monotonically increasing Claude task ID for a list. + /// + /// Keeping the counter beside the projection lets root and child profiles + /// safely create tasks in the same shared list. + pub(crate) fn next_task_id(&self, list_id: &str) -> u64 { + let mut counters = self + .task_counters + .lock() + .expect("task counter lock poisoned"); + let counter = counters.entry(list_id.to_string()).or_default(); + *counter = counter.saturating_add(1); + *counter + } + /// Snapshot the projection for `list_id`. Used by tests and by the /// list-style tools that need a stable view. #[must_use] diff --git a/lib/components/fabro-agent/src/todo_tools.rs b/lib/components/fabro-agent/src/todo_tools.rs index 764bf8bbc..2eb136ed8 100644 --- a/lib/components/fabro-agent/src/todo_tools.rs +++ b/lib/components/fabro-agent/src/todo_tools.rs @@ -10,8 +10,7 @@ use std::collections::{BTreeMap, HashMap, HashSet}; use std::fmt::Write; use std::str::FromStr; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::sync::{Arc, Mutex}; +use std::sync::Arc; use fabro_llm::types::ToolDefinition; use fabro_types::{TodoListKind, TodoProjection, TodoStatus, TodoUpdatedProps}; @@ -395,28 +394,6 @@ pub fn make_todo_list_tool(runtime: Arc) -> RegisteredTool { } } -/// Per-list monotonically-increasing task counter for Anthropic -/// `TaskCreate`. Shared state lives inside the tool closure so two parallel -/// `TaskCreate` calls inside one session can never receive the same ID. -#[derive(Debug, Default)] -struct AnthropicTaskCounters { - counters: Mutex>>, -} - -impl AnthropicTaskCounters { - fn next(&self, list_id: &str) -> u64 { - let counter = { - let mut guard = self.counters.lock().expect("task counter lock poisoned"); - Arc::clone( - guard - .entry(list_id.to_string()) - .or_insert_with(|| Arc::new(AtomicU64::new(0))), - ) - }; - counter.fetch_add(1, Ordering::Relaxed) + 1 - } -} - fn optional_string(args: &Value, key: &str) -> Option { args.get(key) .and_then(Value::as_str) @@ -471,7 +448,6 @@ fn format_task_details(todo: &TodoProjection) -> String { #[must_use] pub fn make_task_create_tool(runtime: Arc) -> RegisteredTool { - let counters = Arc::new(AnthropicTaskCounters::default()); RegisteredTool { definition: ToolDefinition { name: "TaskCreate".into(), @@ -489,7 +465,6 @@ pub fn make_task_create_tool(runtime: Arc) -> RegisteredTool { }, executor: Arc::new(move |args, ctx| { let runtime = runtime.clone(); - let counters = counters.clone(); Box::pin(async move { let list_id = anthropic_task_scope(&ctx)?; let subject = args @@ -502,7 +477,7 @@ pub fn make_task_create_tool(runtime: Arc) -> RegisteredTool { .and_then(Value::as_str) .ok_or_else(|| "Missing required parameter: description".to_string())? .to_string(); - let task_id = counters.next(&list_id); + let task_id = runtime.next_task_id(&list_id); let id_string = task_id.to_string(); let order = u32::try_from(task_id.saturating_sub(1)).unwrap_or(u32::MAX); diff --git a/lib/components/fabro-agent/src/tools.rs b/lib/components/fabro-agent/src/tools.rs index 82deb68e7..a36410b33 100644 --- a/lib/components/fabro-agent/src/tools.rs +++ b/lib/components/fabro-agent/src/tools.rs @@ -656,7 +656,7 @@ fn format_brave_results(body: &serde_json::Value) -> String { output } -fn make_web_search_tool_with_api_key(api_key: String) -> RegisteredTool { +pub(crate) fn make_web_search_tool_with_api_key(api_key: String) -> RegisteredTool { use std::sync::OnceLock; static CLIENT: OnceLock = OnceLock::new(); diff --git a/lib/components/fabro-llm/src/adapter_registry.rs b/lib/components/fabro-llm/src/adapter_registry.rs index 822626f34..26ff6d70b 100644 --- a/lib/components/fabro-llm/src/adapter_registry.rs +++ b/lib/components/fabro-llm/src/adapter_registry.rs @@ -294,14 +294,15 @@ mod tests { #[rustfmt::skip] let expected: &[RouteRow] = &[ // model id deployment_id transport codec billing profile - ("claude-fable-5", "claude-fable-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), + ("claude-fable-5", "claude-fable-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Claude5), ("claude-haiku-4-5", "claude-haiku-4-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), ("claude-opus-4-6", "claude-opus-4-6", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), ("claude-opus-4-7", "claude-opus-4-7", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), ("claude-opus-4-8", "claude-opus-4-8", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), - ("claude-opus-5", "claude-opus-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), + ("claude-opus-5", "claude-opus-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Claude5), ("claude-sonnet-4-5", "claude-sonnet-4-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), ("claude-sonnet-4-6", "claude-sonnet-4-6", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Anthropic), + ("claude-sonnet-5", "claude-sonnet-5", T::Anthropic, C::AnthropicMessages, B::Anthropic, P::Claude5), ("gemini-3-flash-preview", "gemini-3-flash-preview", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini), ("gemini-3.1-flash-lite", "gemini-3.1-flash-lite", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini), ("gemini-3.1-pro-preview", "gemini-3.1-pro-preview", T::Gemini, C::GeminiGenerate, B::Gemini, P::Gemini), @@ -360,7 +361,7 @@ mod tests { let by_alias = resolve_route(catalog, select_from_all(catalog, "sonnet")) .expect("alias should resolve"); - let by_id = resolve_route(catalog, select_from_all(catalog, "claude-sonnet-4-6")) + let by_id = resolve_route(catalog, select_from_all(catalog, "claude-sonnet-5")) .expect("id should resolve"); assert_eq!(by_alias, by_id); diff --git a/lib/components/fabro-workflow/src/handler/llm/api.rs b/lib/components/fabro-workflow/src/handler/llm/api.rs index c4ce3737e..3441aa89b 100644 --- a/lib/components/fabro-workflow/src/handler/llm/api.rs +++ b/lib/components/fabro-workflow/src/handler/llm/api.rs @@ -8,7 +8,7 @@ use fabro_agent::tool_registry::{RegisteredTool, ToolContext, ToolRegistry, Tool use fabro_agent::{ AgentEvent, AgentProfile, AgentProfileBuilder, CompletionCoordinator, Message as AgentMessage, Sandbox, Session, SessionOptions, SessionShutdownReason, StaticEnvProvider, ToolEnvProvider, - ToolSecrets, canonical_tool_name, register_question_tools, + ToolSecrets, WebFetchSummarizer, canonical_tool_name, register_question_tools, }; use fabro_auth::{CredentialSource, EnvCredentialSource}; use fabro_graphviz::graph::{AttrValue, Node}; @@ -19,10 +19,10 @@ use fabro_llm::types::{ }; use fabro_mcp::config::McpServerSettings; #[cfg(test)] -use fabro_model::AgentProfileKind; -#[cfg(test)] use fabro_model::catalog::LlmCatalogSettings; -use fabro_model::{Catalog, FallbackTarget, ModelRef, ProviderId, UsdMicros}; +use fabro_model::{ + AgentProfileKind, Catalog, FallbackTarget, ModelHandle, ModelRef, ProviderId, UsdMicros, +}; use fabro_types::settings::run::RunModelControls; use fabro_types::{PermissionLevel, RunId, SessionCapability, StageId, StageTiming}; use serde::de::DeserializeOwned; @@ -823,6 +823,17 @@ impl AgentApiBackend { Arc::clone(&catalog), ) .with_tool_secrets(tool_secrets); + let profile_builder = if provider.profile_kind == AgentProfileKind::Claude5 { + profile_builder.with_web_fetch_summarizer(Some(WebFetchSummarizer { + client: client.clone(), + model_id: ModelHandle::ByName { + provider: provider.provider_id.clone(), + model: model.to_string(), + }, + })) + } else { + profile_builder + }; let mut profile = profile_builder.build(); let config = SessionOptions { @@ -2843,6 +2854,25 @@ reasoning = false assert_eq!(provider.profile_kind, AgentProfileKind::Anthropic); } + #[test] + fn api_backend_selects_claude5_profile_for_sonnet5() { + let backend = AgentApiBackend::new_with_catalog( + "claude-sonnet-5".to_string(), + ProviderId::anthropic(), + Vec::new(), + Arc::new(EnvCredentialSource::new()), + SteeringHub::for_tests(), + Arc::new(Catalog::from_builtin().unwrap()), + ); + + let provider = backend + .resolve_provider_context("claude-sonnet-5", None) + .unwrap(); + + assert_eq!(provider.provider_id, ProviderId::anthropic()); + assert_eq!(provider.profile_kind, AgentProfileKind::Claude5); + } + #[test] fn api_backend_preserves_default_provider_for_legacy_model_identifier() { let settings: LlmCatalogSettings = toml::from_str( diff --git a/lib/components/fabro-workflow/src/operations/create.rs b/lib/components/fabro-workflow/src/operations/create.rs index 2ec40d443..3f2b739af 100644 --- a/lib/components/fabro-workflow/src/operations/create.rs +++ b/lib/components/fabro-workflow/src/operations/create.rs @@ -1008,7 +1008,7 @@ reasoning = false assert_eq!( validated.graph().nodes["work"].attrs.get("model"), - Some(&AttrValue::String("claude-sonnet-4-6".into())) + Some(&AttrValue::String("claude-sonnet-5".into())) ); } @@ -1416,7 +1416,7 @@ reasoning = false .model .name .as_deref(), - Some("claude-sonnet-4-6") + Some("claude-sonnet-5") ); assert_eq!( created diff --git a/lib/components/fabro-workflow/src/pipeline/transform.rs b/lib/components/fabro-workflow/src/pipeline/transform.rs index b399d6637..45b70792d 100644 --- a/lib/components/fabro-workflow/src/pipeline/transform.rs +++ b/lib/components/fabro-workflow/src/pipeline/transform.rs @@ -157,7 +157,7 @@ mod tests { let transformed = transform(parsed, &transform_options()).unwrap(); assert_eq!( transformed.graph.nodes["work"].attrs.get("model"), - Some(&AttrValue::String("claude-sonnet-4-6".into())) + Some(&AttrValue::String("claude-sonnet-5".into())) ); } @@ -252,7 +252,7 @@ mod tests { ); assert_eq!( lint.attrs.get("model"), - Some(&AttrValue::String("claude-sonnet-4-6".into())) + Some(&AttrValue::String("claude-sonnet-5".into())) ); } diff --git a/lib/components/fabro-workflow/tests/materialize_run.rs b/lib/components/fabro-workflow/tests/materialize_run.rs index 4fec97585..830a73a3b 100644 --- a/lib/components/fabro-workflow/tests/materialize_run.rs +++ b/lib/components/fabro-workflow/tests/materialize_run.rs @@ -40,7 +40,7 @@ fn materialize_run_applies_graph_and_catalog_defaults() { .unwrap(); let resolved = &materialized.run; - assert_eq!(resolved.model.name.as_deref(), Some("claude-sonnet-4-6")); + assert_eq!(resolved.model.name.as_deref(), Some("claude-sonnet-5")); assert_eq!(resolved.model.provider.as_deref(), Some("anthropic")); assert_eq!( materialized.run.goal.as_ref(), diff --git a/lib/foundation/fabro-model/src/adapter.rs b/lib/foundation/fabro-model/src/adapter.rs index 5ff264b4c..c9eb25b97 100644 --- a/lib/foundation/fabro-model/src/adapter.rs +++ b/lib/foundation/fabro-model/src/adapter.rs @@ -67,6 +67,12 @@ impl AsRef for AdapterKind { #[strum(serialize_all = "snake_case")] pub enum AgentProfileKind { Anthropic, + /// Claude 5 models trained against Anthropic's current coding-agent + /// harness. This remains model-scoped so older Claude models keep the + /// established Anthropic profile. + #[serde(rename = "claude-5")] + #[strum(to_string = "claude-5")] + Claude5, #[serde(rename = "openai")] #[strum(to_string = "openai")] OpenAi, diff --git a/lib/foundation/fabro-model/src/catalog.rs b/lib/foundation/fabro-model/src/catalog.rs index f012a62ff..d020bfdba 100644 --- a/lib/foundation/fabro-model/src/catalog.rs +++ b/lib/foundation/fabro-model/src/catalog.rs @@ -2846,7 +2846,7 @@ enabled = true catalog .default_for_provider(&bedrock) .map(|model| model.id.as_str()), - Some("claude-sonnet-4-6") + Some("claude-sonnet-5") ); // Fable 5 ships with sampling params pinned off (the Converse // encoder drops temperature/top_p for it). @@ -2854,12 +2854,11 @@ enabled = true .get_on_provider(&bedrock, "claude-fable-5") .expect("fable row should be present"); assert!(!fable.features.sampling_params); - assert!( - catalog - .settings_for(fable) - .expect("fable settings should be present") - .reasoning_by_default - ); + let fable_settings = catalog + .settings_for(fable) + .expect("fable settings should be present"); + assert!(fable_settings.reasoning_by_default); + assert_eq!(fable_settings.agent_profile, AgentProfileKind::Claude5); assert_eq!( catalog .model_settings_on_provider(&bedrock, "claude-fable-5") @@ -2867,6 +2866,16 @@ enabled = true .billing_policy, BillingPolicy::Anthropic ); + let sonnet = catalog + .get_on_provider(&bedrock, "claude-sonnet-5") + .expect("Sonnet 5 row should be present"); + assert_eq!(sonnet.limits.context_window, 1_000_000); + assert_eq!(sonnet.limits.max_output, Some(128_000)); + assert!(!sonnet.features.sampling_params); + assert_eq!( + catalog.settings_for(sonnet).unwrap().agent_profile, + AgentProfileKind::Claude5 + ); } #[test] @@ -3013,7 +3022,7 @@ enabled = true // open-weights rows inherit it. assert_eq!( catalog - .model_settings_on_provider(&openrouter, "claude-sonnet-4-6") + .model_settings_on_provider(&openrouter, "claude-sonnet-5") .unwrap() .billing_policy, BillingPolicy::Anthropic @@ -3029,7 +3038,7 @@ enabled = true catalog .default_for_provider(&openrouter) .map(|model| model.id.as_str()), - Some("claude-sonnet-4-6") + Some("claude-sonnet-5") ); } @@ -3122,6 +3131,19 @@ enabled = true true, BillingPolicy::Anthropic, ), + ( + "claude-sonnet-5", + "anthropic/claude-sonnet-5", + "claude-5", + 1_000_000, + 2.0, + 10.0, + 0.2, + ReasoningEffortFeature::Levels, + false, + true, + BillingPolicy::Anthropic, + ), ]; for ( @@ -3173,13 +3195,21 @@ enabled = true ReasoningEffort::VARIANTS, "{id}" ); + if family == "claude-5" { + assert_eq!(settings.agent_profile, AgentProfileKind::Claude5, "{id}"); + } } - for alias in ["opus", "claude-opus"] { + for (alias, expected) in [ + ("opus", "claude-opus-5"), + ("claude-opus", "claude-opus-5"), + ("sonnet", "claude-sonnet-5"), + ("claude-sonnet", "claude-sonnet-5"), + ] { let model = catalog .resolve_on_provider(&ProviderId::new("openrouter"), alias) .unwrap_or_else(|error| panic!("{alias} should resolve on OpenRouter: {error}")); - assert_eq!(model.id, "claude-opus-5", "{alias}"); + assert_eq!(model.id, expected, "{alias}"); } } @@ -3868,7 +3898,7 @@ enabled = true let m = Catalog::builtin() .default_for_provider(&ProviderId::anthropic()) .unwrap(); - assert_eq!(m.id, "claude-sonnet-4-6"); + assert_eq!(m.id, "claude-sonnet-5"); assert!(m.default); let m = Catalog::builtin() diff --git a/lib/foundation/fabro-model/src/catalog/providers/anthropic.toml b/lib/foundation/fabro-model/src/catalog/providers/anthropic.toml index 25e9774ae..6d0adb16c 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/anthropic.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/anthropic.toml @@ -13,6 +13,7 @@ header = { custom = "x-api-key" } display_name = "Claude Fable 5" family = "claude-5" aliases = ["fable", "claude-fable"] +agent_profile = "claude-5" [providers.anthropic.models."claude-fable-5".limits] context_window = 1000000 @@ -37,6 +38,7 @@ family = "claude-5" training = "2026-05-01" knowledge_cutoff = "May 2026" aliases = ["opus", "claude-opus"] +agent_profile = "claude-5" [providers.anthropic.models."claude-opus-5".limits] context_window = 1000000 @@ -63,6 +65,33 @@ input_cost_per_mtok = 10.0 output_cost_per_mtok = 50.0 cache_input_cost_per_mtok = 1.0 +[providers.anthropic.models."claude-sonnet-5"] +display_name = "Claude Sonnet 5" +family = "claude-5" +training = "2026-01-01" +knowledge_cutoff = "Jan 2026" +default = true +aliases = ["sonnet", "claude-sonnet"] +agent_profile = "claude-5" + +[providers.anthropic.models."claude-sonnet-5".limits] +context_window = 1000000 +max_output = 128000 + +[providers.anthropic.models."claude-sonnet-5".features] +tools = true +vision = true +reasoning = true +reasoning_effort = "levels" +prompt_cache = true +sampling_params = false + +# Introductory pricing through August 31, 2026. +[providers.anthropic.models."claude-sonnet-5".costs] +input_cost_per_mtok = 2.0 +output_cost_per_mtok = 10.0 +cache_input_cost_per_mtok = 0.2 + [providers.anthropic.models."claude-opus-4-8"] display_name = "Claude Opus 4.8" family = "claude-4" @@ -188,9 +217,7 @@ display_name = "Claude Sonnet 4.6" family = "claude-4" training = "2025-08-01" knowledge_cutoff = "May 2025" -default = true estimated_output_tps = 50 -aliases = ["sonnet", "claude-sonnet"] [providers.anthropic.models."claude-sonnet-4-6".limits] context_window = 200000 diff --git a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml index 71c74d686..401037c42 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml @@ -41,16 +41,15 @@ credentials = [ # ---------- Anthropic Claude ---------- # # Claude bills Anthropic-style cache reads/writes, so these rows override -# the provider's billing default. Claude Fable 5 appears at the end of this -# file because its Bedrock deployment pins sampling parameters and requires an -# extra data-sharing opt-in. +# the provider's billing default. Claude 5 models appear at the end of this +# file because their Bedrock deployments pin sampling parameters and require +# extra endpoint-specific handling. [providers.bedrock.models."claude-sonnet-4-6"] api_id = "us.anthropic.claude-sonnet-4-6" display_name = "Claude Sonnet 4.6 (Bedrock)" family = "claude-4" billing_policy = "anthropic" -default = true [providers.bedrock.models."claude-sonnet-4-6".limits] context_window = 1000000 @@ -360,6 +359,7 @@ api_id = "us.anthropic.claude-fable-5" display_name = "Claude Fable 5 (Bedrock)" family = "claude-5" billing_policy = "anthropic" +agent_profile = "claude-5" [providers.bedrock.models."claude-fable-5".limits] context_window = 1000000 @@ -377,3 +377,33 @@ sampling_params = false input_cost_per_mtok = 10.0 output_cost_per_mtok = 50.0 cache_input_cost_per_mtok = 1.0 + +# Claude Sonnet 5 uses adaptive thinking by default and rejects non-default +# sampling parameters. Effort-level mapping through +# additionalModelRequestFields is a named follow-up, as for Fable 5. + +[providers.bedrock.models."claude-sonnet-5"] +api_id = "us.anthropic.claude-sonnet-5" +display_name = "Claude Sonnet 5 (Bedrock)" +family = "claude-5" +billing_policy = "anthropic" +default = true +agent_profile = "claude-5" + +[providers.bedrock.models."claude-sonnet-5".limits] +context_window = 1000000 +max_output = 128000 + +[providers.bedrock.models."claude-sonnet-5".features] +tools = true +vision = true +reasoning = true +reasoning_by_default = true +prompt_cache = true +sampling_params = false + +# Introductory pricing through August 31, 2026. +[providers.bedrock.models."claude-sonnet-5".costs] +input_cost_per_mtok = 2.0 +output_cost_per_mtok = 10.0 +cache_input_cost_per_mtok = 0.2 diff --git a/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml b/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml index 294dea6eb..f57361b0f 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/openrouter.toml @@ -39,6 +39,7 @@ display_name = "Claude Fable 5 (via OpenRouter)" family = "claude-5" billing_policy = "anthropic" aliases = ["fable", "claude-fable"] +agent_profile = "claude-5" [providers.openrouter.models."claude-fable-5".limits] context_window = 1000000 @@ -66,6 +67,7 @@ billing_policy = "anthropic" training = "2026-05-01" knowledge_cutoff = "May 2026" aliases = ["opus", "claude-opus"] +agent_profile = "claude-5" [providers.openrouter.models."claude-opus-5".limits] context_window = 1000000 @@ -85,6 +87,37 @@ input_cost_per_mtok = 5.0 output_cost_per_mtok = 25.0 cache_input_cost_per_mtok = 0.5 +[providers.openrouter.models."claude-sonnet-5"] +api_id = "anthropic/claude-sonnet-5" +display_name = "Claude Sonnet 5 (via OpenRouter)" +family = "claude-5" +billing_policy = "anthropic" +training = "2026-01-01" +knowledge_cutoff = "Jan 2026" +default = true +aliases = ["sonnet", "claude-sonnet"] +agent_profile = "claude-5" + +[providers.openrouter.models."claude-sonnet-5".limits] +context_window = 1000000 +max_output = 128000 + +[providers.openrouter.models."claude-sonnet-5".features] +tools = true +vision = true +reasoning = true +reasoning_effort = "levels" +prompt_cache = true +cache_control_breakpoints = true +sampling_params = false + +# Current introductory rate. OpenRouter's authoritative in-band usage.cost +# supersedes this estimate on completed responses. +[providers.openrouter.models."claude-sonnet-5".costs] +input_cost_per_mtok = 2.0 +output_cost_per_mtok = 10.0 +cache_input_cost_per_mtok = 0.2 + [providers.openrouter.models."claude-opus-4-8"] api_id = "anthropic/claude-opus-4.8" display_name = "Claude Opus 4.8 (via OpenRouter)" @@ -138,8 +171,6 @@ api_id = "anthropic/claude-sonnet-4.6" display_name = "Claude Sonnet 4.6 (via OpenRouter)" family = "claude-4" billing_policy = "anthropic" -default = true -aliases = ["sonnet", "claude-sonnet"] [providers.openrouter.models."claude-sonnet-4-6".limits] context_window = 1000000 From c4971b93d32a950bb971fa3c5902e083c8905b2e Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 14:54:38 -0400 Subject: [PATCH 02/76] fix(timing): accumulate active time for in-flight stages MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `active_time_ms` was only ever computed from terminal stage events, so a stage still running contributed zero to the run rollup. A run parked in one long agent stage reported 2m 8s of active time against 16m 53s of wall clock — the two finished stages — while the running stage had been doing continuous inference and tool work for over 14 minutes. `live_run_timing` summed `filter_map(|stage| stage.timing)`, and `stage.timing` is only written at finalization. Wall time ticked live off `start_time`; active time did not tick at all. Stage projections now accumulate brackets from the event log: - Closing an inference bracket folds its span into `live_inference_ms` instead of discarding it, including across retries, matching the in-process stopwatch. - Tool calls open a batch on the first outstanding call and close it when the last one drains, so tools running concurrently within a turn count once — the same span `execute_tool_calls` is bracketed by. Summing per-call durations would over-count parallel tool use. Subagent tool events are excluded; they run inside the root call's span already. - `StageProjection::live_timing(now)` composes accumulators with any open bracket, per handler: agent stages use the brackets, prompt and command stages count elapsed time as inference and tool respectively, and handlers that wait on a human, timer, condition, or child branches report zero. Active is clamped to wall per stage. A worker killed mid-turn leaves its bracket open forever, and without the clamp it would tick up unbounded. The clamp does not need to detect the dead worker: a stage cannot have been active longer than it has existed. `watchdog.timeout` remains the authority on whether a run is stuck. The clamp is deliberately not applied at run level, where concurrent branches can legitimately sum past run wall time. Timing is derived from events rather than emitted by the worker, so this needs no event-schema change and applies to runs already stored. `StageProjection.timing` keeps its terminal-only meaning, and the authoritative breakdown still replaces the live estimate at terminal events. The billing endpoint had the same hole behind its `wall_only` fallback: running stages reported zero inference/tool/active. Not visible in the product, which renders only `wall_time_ms`, but wrong for any other consumer of `GET /runs/{id}/billing`. Parallel branch stages lose their breakdown permanently, even after completion, because `parallel.branch.completed` carries only `duration_ms`. That is a separate data-loss bug, tracked in #644. Co-Authored-By: Claude Opus 5 (1M context) --- .../app/routes/run-detail/header.tsx | 9 +- ...05-21-wall-and-active-time-metrics-plan.md | 11 +- docs/public/api-reference/fabro-api.yaml | 66 ++- .../src/server/handler/billing.rs | 16 +- lib/components/fabro-store/src/run_state.rs | 475 +++++++++++++++++- .../fabro-types/src/run_projection.rs | 462 ++++++++++++++++- lib/foundation/fabro-types/src/timing.rs | 26 + .../src/.openapi-generator/FILES | 1 + .../fabro-api-client/src/models/index.ts | 1 + .../fabro-api-client/src/models/run-timing.ts | 2 +- .../src/models/stage-projection.ts | 12 + .../src/models/stage-timing.ts | 2 +- .../src/models/stage-tool-batch-projection.ts | 29 ++ 13 files changed, 1060 insertions(+), 52 deletions(-) create mode 100644 lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts diff --git a/apps/fabro-web/app/routes/run-detail/header.tsx b/apps/fabro-web/app/routes/run-detail/header.tsx index efc68f09b..3b7081596 100644 --- a/apps/fabro-web/app/routes/run-detail/header.tsx +++ b/apps/fabro-web/app/routes/run-detail/header.tsx @@ -326,6 +326,7 @@ function DurationPopover({ }) { const endMs = completedAt != null ? Date.parse(completedAt) : now; const sinceCreatedMs = Math.max(0, endMs - Date.parse(createdAt)); + const isRunning = completedAt == null; return ( <> Duration @@ -335,8 +336,14 @@ function DurationPopover({
{formatDurationMs(sinceCreatedMs)}
-
Active (inference + tools)
+
+ Active (inference + tools){isRunning ? " — estimated" : ""} +
{formatDurationMs(timing.active_time_ms)}
+
+ {formatDurationMs(timing.inference_time_ms)} inference ·{" "} + {formatDurationMs(timing.tool_time_ms)} tools +
diff --git a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md index 90438d3b5..3683e5879 100644 --- a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md +++ b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md @@ -157,7 +157,14 @@ git diff --check provider-reported model-only compute time. - LLM retry backoff, queueing outside a request/stream, human waits, steering waits, and scheduler gaps are wall time but not active time. -- Active timing is finalized-event based in v1; live active-time ticking can be - added later if it becomes necessary. +- ~~Active timing is finalized-event based in v1; live active-time ticking can + be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became + necessary: a run parked in one long agent stage reported ~12% of its wall time + as active, because in-flight stages contributed nothing. Stage projections now + accumulate inference and tool brackets from the event log and expose + `StageProjection::live_timing(now)`, the active-time twin of + `live_wall_time_ms`. Finalized values remain authoritative and still replace + the live estimate at terminal events. See + `.ai/plans/live-active-time-accumulation.md`. - No compatibility layer is required for existing API clients or stored run event data. diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml index b028d89c0..568aa30e5 100644 --- a/docs/public/api-reference/fabro-api.yaml +++ b/docs/public/api-reference/fabro-api.yaml @@ -10673,7 +10673,38 @@ components: - type: "null" description: | Per-attempt timing breakdown for the latest terminal attempt: - wall time plus the active inference/tool breakdown. + wall time plus the active inference/tool breakdown. Null while the + stage is still in flight; the live estimate is derived from + `live_inference_ms`, `live_tool_ms`, and any open bracket. + live_inference_ms: + type: integer + format: uint64 + minimum: 0 + default: 0 + description: | + Inference time accumulated from closed brackets during the current + attempt. Live estimate only — the authoritative value arrives with + the terminal event and lands in `timing`. Excludes the currently + open bracket, whose span is measured from `inference.started_at`. + example: 78230 + live_tool_ms: + type: integer + format: uint64 + minimum: 0 + default: 0 + description: | + Tool time accumulated from closed tool batches during the current + attempt. A batch spans the first dispatched call through the + completion that drains the last outstanding one, so tools running + concurrently within a turn are counted once. + example: 7588 + tool_batch: + oneOf: + - $ref: "#/components/schemas/StageToolBatchProjection" + - type: "null" + description: | + Open tool batch: when the batch started and which calls have not + yet reported completion. usage: $ref: "#/components/schemas/BilledTokenCounts" model: @@ -10734,6 +10765,28 @@ components: $ref: "#/components/schemas/StageState" description: Lifecycle state of the stage projection. + StageToolBatchProjection: + description: > + One open tool batch: tool calls dispatched together that have not all + reported completion. `open_call_ids` is a set rather than a count so a + duplicated completion in a replayed log cannot drain the batch early. + type: object + required: + - started_at + - open_call_ids + properties: + started_at: + type: string + format: date-time + description: > + When the batch opened — the first dispatched call observed while no + other calls were outstanding. + open_call_ids: + type: array + items: + type: string + description: Calls dispatched but not yet completed, by tool call id. + StageInferenceProjection: description: > One open inference bracket: a dispatched LLM request that has not yet @@ -12244,6 +12297,12 @@ components: observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. + + For a terminal stage these come from the worker's own stopwatch and are + authoritative. For a stage still in flight they are a live estimate + reconstructed from the event log, and `active_time_ms` is clamped to + `wall_time_ms`. The estimate is replaced by the authoritative + breakdown when the stage reaches a terminal event. type: object required: - wall_time_ms @@ -12278,6 +12337,11 @@ components: Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. + + For a running run, stages still in flight contribute a live estimate + rather than nothing, so wall and active both advance continuously. + Unlike `StageTiming`, active is not clamped to wall here — concurrent + branches can legitimately sum past run wall time. type: object required: - wall_time_ms diff --git a/lib/apps/fabro-server/src/server/handler/billing.rs b/lib/apps/fabro-server/src/server/handler/billing.rs index a1888713c..6cc4cbdde 100644 --- a/lib/apps/fabro-server/src/server/handler/billing.rs +++ b/lib/apps/fabro-server/src/server/handler/billing.rs @@ -175,7 +175,7 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec= row.latest_visit { @@ -188,20 +188,6 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec) -> StageTiming { - if let Some(timing) = stage.timing { - return timing; - } - if let Some(live_wall) = stage.live_wall_time_ms(now) { - return StageTiming::wall_only(live_wall); - } - StageTiming::default() -} - fn stage_has_billing_row(stage: &StageProjection) -> bool { stage.completion.is_some() || stage.timing.is_some() diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs index 9ab1e2b2d..df05f84b3 100644 --- a/lib/components/fabro-store/src/run_state.rs +++ b/lib/components/fabro-store/src/run_state.rs @@ -449,7 +449,7 @@ impl RunProjectionReducer for RunProjection { context_window.event_seq = Some(event.seq); stage.context_window = Some(context_window); } - close_inference_bracket(self, stored, props.visit, event.seq); + close_inference_bracket(self, stored, props.visit, event.seq, ts); } EventBody::AgentLlmStarted(props) => { open_inference_bracket(self, stored, props, event.seq, ts); @@ -478,10 +478,10 @@ impl RunProjectionReducer for RunProjection { inference.first_output_kind = None; } EventBody::AgentError(props) => { - close_inference_bracket(self, stored, props.visit, event.seq); + close_inference_bracket(self, stored, props.visit, event.seq, ts); } EventBody::AgentSessionEnded(_) => { - close_inference_brackets_for_session(self, stored); + close_inference_brackets_for_session(self, stored, ts); } EventBody::AgentSessionActivated(props) => { let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) @@ -497,7 +497,7 @@ impl RunProjectionReducer for RunProjection { return Ok(()); }; stage.agent_control = AgentControlState::WaitingForSteer; - close_inference_bracket(self, stored, props.visit, event.seq); + close_inference_bracket(self, stored, props.visit, event.seq, ts); } EventBody::AgentSteeringInjected(props) => { let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) @@ -746,6 +746,7 @@ impl RunProjectionReducer for RunProjection { }); } EventBody::AgentToolStarted(props) => { + let is_root_session = stored.parent_session_id.is_none(); let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) else { return Ok(()); @@ -766,6 +767,22 @@ impl RunProjectionReducer for RunProjection { projection.invoked = true; } } + // A subagent's tools run inside the root session's tool call, + // so the root batch already covers them. Timing them again + // would double-count that span. + if is_root_session { + stage.open_tool_call(props.tool_call_id.clone(), ts); + } + } + EventBody::AgentToolCompleted(props) => { + if stored.parent_session_id.is_some() { + return Ok(()); + } + let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) + else { + return Ok(()); + }; + stage.close_tool_call(&props.tool_call_id, ts); } _ => {} } @@ -1044,6 +1061,22 @@ fn matching_inference_slot<'a>( visit: u32, seq: u32, ) -> Option<&'a mut Option> { + Some(&mut matching_inference_stage(state, stored, visit, seq)?.inference) +} + +/// Resolve the stage owning a bracket this event is allowed to mutate, +/// borrowing the whole projection so the caller can also fold elapsed time +/// into the stage's live accumulators. +/// +/// Same gating as [`matching_inference_slot`]: `None` for child-session +/// events, for a stage with no open bracket, and for a bracket belonging to a +/// different session (which is what post-failover events look like). +fn matching_inference_stage<'a>( + state: &'a mut RunProjection, + stored: &RunEvent, + visit: u32, + seq: u32, +) -> Option<&'a mut StageProjection> { if stored.parent_session_id.is_some() { return None; } @@ -1053,7 +1086,7 @@ fn matching_inference_slot<'a>( .inference .as_ref() .is_some_and(|inference| inference.session_id == session_id); - opened_here.then_some(&mut stage.inference) + opened_here.then_some(stage) } /// Resolve the open inference bracket this event is allowed to mutate. @@ -1066,11 +1099,37 @@ fn matching_inference_bracket<'a>( matching_inference_slot(state, stored, visit, seq)?.as_mut() } -/// Close the bracket on a stage-addressed terminal event. -fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: u32, seq: u32) { - if let Some(slot) = matching_inference_slot(state, stored, visit, seq) { - *slot = None; - } +/// Close the bracket on a stage-addressed terminal event, folding its elapsed +/// time into the stage's live inference accumulator. +fn close_inference_bracket( + state: &mut RunProjection, + stored: &RunEvent, + visit: u32, + seq: u32, + ts: DateTime, +) { + let Some(stage) = matching_inference_stage(state, stored, visit, seq) else { + return; + }; + close_bracket_on_stage(stage, ts); +} + +/// Take the open bracket and add its span to `live_inference_ms`. +/// +/// Retries inside the bracket are deliberately included: the in-process +/// stopwatch counts a retried attempt's elapsed time as inference, and +/// `agent.llm.retry` keeps the bracket open rather than reopening it. +fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime) { + let Some(inference) = stage.inference.take() else { + return; + }; + stage.accumulate_inference_ms(elapsed_ms(inference.started_at, ts)); +} + +/// Non-negative milliseconds between two instants, saturating at zero so a +/// clock skew or an out-of-order replay cannot produce a negative span. +fn elapsed_ms(from: DateTime, to: DateTime) -> u64 { + u64::try_from(to.signed_duration_since(from).num_milliseconds().max(0)).unwrap_or(0) } /// Close every bracket opened by the session that just ended. @@ -1088,7 +1147,11 @@ fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: /// opened. Implemented as a normal stage lookup it would find no target and /// silently no-op, leaving the bracket open forever on exactly the path it /// exists to cover. -fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunEvent) { +fn close_inference_brackets_for_session( + state: &mut RunProjection, + stored: &RunEvent, + ts: DateTime, +) { if stored.parent_session_id.is_some() { return; } @@ -1101,7 +1164,7 @@ fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunE .as_ref() .is_some_and(|inference| inference.session_id == session_id); if opened_here { - stage.inference = None; + close_bracket_on_stage(stage, ts); } } } @@ -1417,18 +1480,17 @@ fn finalize_unfinished_stages_after_run_failed( continue; } + // Close any bracket still open so its span is not dropped on the + // floor when the live estimate is frozen into `timing` below. + close_bracket_on_stage(stage, timestamp); + + // Freeze the live estimate before flipping to a terminal state: + // `live_timing` reads `effective_state` and would return wall-only + // once the stage no longer looks in-flight. + let frozen = stage.live_timing(timestamp); stage.state = terminal_state; - if stage.timing.is_none() { - if let Some(started_at) = stage.started_at { - let wall_time_ms = u64::try_from( - timestamp - .signed_duration_since(started_at) - .num_milliseconds() - .max(0), - ) - .expect("non-negative milliseconds fit in u64"); - stage.timing = Some(fabro_types::StageTiming::wall_only(wall_time_ms)); - } + if stage.timing.is_none() && stage.started_at.is_some() { + stage.timing = Some(frozen); } } } @@ -1560,6 +1622,373 @@ mod tests { use super::{RunProjection, RunProjectionReducer, build_summary}; use crate::{Error, EventEnvelope, StageId}; + /// Live accumulation of inference and tool time while a stage is in + /// flight. The finalized breakdown still arrives with the terminal event + /// and replaces these; these exist so a long-running stage is not reported + /// as doing no work. + mod live_active_accumulation { + use fabro_types::run_event::{ + AgentLlmFirstOutputProps, AgentLlmRetryProps, AgentLlmStartedProps, + AgentToolCompletedProps, AgentToolStartedProps, + }; + use fabro_types::{ + LlmOutputKind, LlmRetryPhase, ModelRef, Speed, StageOutcome, StageProjection, + }; + + use super::*; + + fn stage_id() -> StageId { + StageId::new("plan", 1) + } + + fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope { + let mut event = test_stage_event_at(seq, ts, body, stage_id()); + event.event.session_id = Some("session-1".to_string()); + event + } + + /// An event from a sub-agent session nested under the root session. + fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope { + let mut event = agent_event(seq, ts, body); + event.event.session_id = Some("session-child".to_string()); + event.event.parent_session_id = Some("session-1".to_string()); + event + } + + fn llm_started() -> EventBody { + EventBody::AgentLlmStarted(AgentLlmStartedProps { + requested_model: ModelRef { + provider: "anthropic".parse().unwrap(), + model_id: "claude-fable-5".into(), + speed: Some(Speed::Fast), + }, + visit: 1, + }) + } + + fn tool_started(tool_call_id: &str) -> EventBody { + EventBody::AgentToolStarted(AgentToolStartedProps { + tool_name: "Bash".to_string(), + tool_call_id: tool_call_id.to_string(), + arguments: json!({}), + visit: 1, + tool_call: None, + turn_id: None, + parent_message_id: None, + }) + } + + fn tool_completed(tool_call_id: &str) -> EventBody { + EventBody::AgentToolCompleted(AgentToolCompletedProps { + tool_name: "Bash".to_string(), + tool_call_id: tool_call_id.to_string(), + output: json!("ok"), + is_error: false, + visit: 1, + tool_result: None, + turn_id: None, + }) + } + + fn agent_message() -> EventBody { + EventBody::AgentMessage(live_agent_message_props(live_counts(10, 5))) + } + + fn started_state() -> RunProjection { + let mut state = initialized_projection(); + state + .apply_event(&test_stage_event_at( + 1, + "2026-04-07T12:00:00Z", + EventBody::StageStarted(started_props()), + stage_id(), + )) + .unwrap(); + state + } + + fn stage(state: &RunProjection) -> &StageProjection { + state.stage(&stage_id()).unwrap() + } + + #[test] + fn closing_an_inference_bracket_accumulates_its_span() { + let mut state = started_state(); + state + .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started())) + .unwrap(); + state + .apply_event(&agent_event( + 3, + "2026-04-07T12:00:06Z", + EventBody::AgentLlmFirstOutput(AgentLlmFirstOutputProps { + kind: LlmOutputKind::Text, + visit: 1, + }), + )) + .unwrap(); + state + .apply_event(&agent_event(4, "2026-04-07T12:00:12Z", agent_message())) + .unwrap(); + + // 12:00:05 -> 12:00:12; first_output is a marker, not the close. + assert_eq!(stage(&state).live_inference_ms, 7_000); + assert!(stage(&state).inference.is_none()); + } + + #[test] + fn concurrent_tool_calls_count_once_not_per_call() { + let mut state = started_state(); + for (seq, id) in [(2, "call-a"), (3, "call-b"), (4, "call-c")] { + state + .apply_event(&agent_event(seq, "2026-04-07T12:00:00Z", tool_started(id))) + .unwrap(); + } + // All three finish 10s later. Summing per-call spans would report + // 30s; the batch actually occupied 10s of wall time. + for (seq, id) in [(5, "call-a"), (6, "call-b"), (7, "call-c")] { + state + .apply_event(&agent_event( + seq, + "2026-04-07T12:00:10Z", + tool_completed(id), + )) + .unwrap(); + } + + assert_eq!(stage(&state).live_tool_ms, 10_000); + assert!(stage(&state).tool_batch.is_none()); + } + + #[test] + fn a_batch_stays_open_until_its_last_call_reports() { + let mut state = started_state(); + state + .apply_event(&agent_event( + 2, + "2026-04-07T12:00:00Z", + tool_started("call-a"), + )) + .unwrap(); + state + .apply_event(&agent_event( + 3, + "2026-04-07T12:00:02Z", + tool_started("call-b"), + )) + .unwrap(); + state + .apply_event(&agent_event( + 4, + "2026-04-07T12:00:05Z", + tool_completed("call-a"), + )) + .unwrap(); + + assert_eq!( + stage(&state).live_tool_ms, + 0, + "batch must not close while call-b is outstanding" + ); + + state + .apply_event(&agent_event( + 5, + "2026-04-07T12:00:09Z", + tool_completed("call-b"), + )) + .unwrap(); + + // Measured from the batch open, not from the last call's start. + assert_eq!(stage(&state).live_tool_ms, 9_000); + } + + #[test] + fn successive_batches_accumulate() { + let mut state = started_state(); + for (seq, ts, body) in [ + (2, "2026-04-07T12:00:00Z", tool_started("call-a")), + (3, "2026-04-07T12:00:04Z", tool_completed("call-a")), + (4, "2026-04-07T12:00:10Z", tool_started("call-b")), + (5, "2026-04-07T12:00:16Z", tool_completed("call-b")), + ] { + state.apply_event(&agent_event(seq, ts, body)).unwrap(); + } + + assert_eq!(stage(&state).live_tool_ms, 10_000); + } + + #[test] + fn a_duplicate_completion_does_not_drain_the_batch_early() { + let mut state = started_state(); + state + .apply_event(&agent_event( + 2, + "2026-04-07T12:00:00Z", + tool_started("call-a"), + )) + .unwrap(); + state + .apply_event(&agent_event( + 3, + "2026-04-07T12:00:00Z", + tool_started("call-b"), + )) + .unwrap(); + // call-a reports twice, as a replayed or duplicated log can. + state + .apply_event(&agent_event( + 4, + "2026-04-07T12:00:03Z", + tool_completed("call-a"), + )) + .unwrap(); + state + .apply_event(&agent_event( + 5, + "2026-04-07T12:00:04Z", + tool_completed("call-a"), + )) + .unwrap(); + + assert_eq!(stage(&state).live_tool_ms, 0); + assert!(stage(&state).tool_batch.is_some()); + } + + #[test] + fn subagent_tool_calls_do_not_double_count_against_the_root_batch() { + let mut state = started_state(); + state + .apply_event(&agent_event( + 2, + "2026-04-07T12:00:00Z", + tool_started("root-call"), + )) + .unwrap(); + // The sub-agent's own tools run inside the root call's span. + state + .apply_event(&child_event( + 3, + "2026-04-07T12:00:01Z", + tool_started("child-call"), + )) + .unwrap(); + state + .apply_event(&child_event( + 4, + "2026-04-07T12:00:02Z", + tool_completed("child-call"), + )) + .unwrap(); + state + .apply_event(&agent_event( + 5, + "2026-04-07T12:00:08Z", + tool_completed("root-call"), + )) + .unwrap(); + + assert_eq!(stage(&state).live_tool_ms, 8_000); + } + + #[test] + fn session_end_accumulates_rather_than_discarding_the_bracket() { + let mut state = started_state(); + state + .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started())) + .unwrap(); + + let mut ended = test_stage_event_at( + 3, + "2026-04-07T12:00:20Z", + EventBody::AgentSessionEnded(AgentSessionEndedProps {}), + stage_id(), + ); + ended.event.session_id = Some("session-1".to_string()); + state.apply_event(&ended).unwrap(); + + assert_eq!(stage(&state).live_inference_ms, 15_000); + assert!(stage(&state).inference.is_none()); + } + + #[test] + fn a_foreign_session_close_leaves_the_bracket_and_accumulator_alone() { + let mut state = started_state(); + state + .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started())) + .unwrap(); + + // Post-failover: a new session emits the message, so the old + // bracket is not this event's to close or bill. + let mut foreign = + test_stage_event_at(3, "2026-04-07T12:00:20Z", agent_message(), stage_id()); + foreign.event.session_id = Some("session-2".to_string()); + state.apply_event(&foreign).unwrap(); + + assert_eq!(stage(&state).live_inference_ms, 0); + assert!(stage(&state).inference.is_some()); + } + + #[test] + fn a_retry_keeps_accumulating_within_one_bracket() { + let mut state = started_state(); + state + .apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started())) + .unwrap(); + state + .apply_event(&agent_event( + 3, + "2026-04-07T12:00:04Z", + EventBody::AgentLlmRetry(AgentLlmRetryProps { + provider: "anthropic".to_string(), + model: "claude-fable-5".to_string(), + attempt: 0, + delay_secs: 0.0, + error: json!({ "kind": "stream" }), + phase: Some(LlmRetryPhase::Consume), + visit: 1, + }), + )) + .unwrap(); + state + .apply_event(&agent_event(4, "2026-04-07T12:00:11Z", agent_message())) + .unwrap(); + + // The whole bracket counts, retry included, matching the + // in-process stopwatch. + assert_eq!(stage(&state).live_inference_ms, 11_000); + assert_eq!(stage(&state).inference, None); + } + + #[test] + fn stage_completion_replaces_the_live_estimate_with_finalized_timing() { + let mut state = started_state(); + state + .apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started())) + .unwrap(); + state + .apply_event(&agent_event(3, "2026-04-07T12:00:09Z", agent_message())) + .unwrap(); + assert_eq!(stage(&state).live_inference_ms, 9_000); + + state + .apply_event(&test_stage_event_at( + 4, + "2026-04-07T12:00:10Z", + EventBody::StageCompleted(completed_props(10_000, StageOutcome::Succeeded)), + stage_id(), + )) + .unwrap(); + + let stage = stage(&state); + assert_eq!( + stage.live_timing(test_dt("2026-04-07T12:30:00Z")), + stage.timing.unwrap(), + "a terminal stage reports its finalized breakdown, not a live estimate" + ); + } + } + fn test_event(seq: u32, body: EventBody, node_id: Option<&str>) -> EventEnvelope { let event = RunEvent { id: format!("evt-{seq}"), diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs index 027cce2fc..0576d32f6 100644 --- a/lib/foundation/fabro-types/src/run_projection.rs +++ b/lib/foundation/fabro-types/src/run_projection.rs @@ -1,5 +1,5 @@ use std::borrow::Cow; -use std::collections::{BTreeMap, HashMap}; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use std::num::NonZeroU32; use chrono::{DateTime, Utc}; @@ -354,10 +354,31 @@ pub struct StageProjection { /// immutable projections under their own `StageId`s. /// /// `None` for stages still in flight (`started_at` is set but no terminal - /// event has been observed yet). For live wall-time ticking, the UI uses - /// `started_at`; once terminal this carries the finalized breakdown. + /// event has been observed yet). For a live breakdown while in flight, use + /// [`StageProjection::live_timing`]; once terminal this carries the + /// finalized, authoritative breakdown. #[serde(default, skip_serializing_if = "Option::is_none")] pub timing: Option, + /// Inference time accumulated from closed brackets during this attempt. + /// + /// Live estimate only: the authoritative value arrives with the terminal + /// event and lands in `timing`. Excludes the currently-open bracket, which + /// [`StageProjection::live_timing`] adds from `inference.started_at`. + #[serde(default, skip_serializing_if = "is_zero_ms")] + pub live_inference_ms: u64, + /// Tool time accumulated from closed tool batches during this attempt. + /// + /// A batch spans the first `agent.tool.started` with no outstanding calls + /// through the `agent.tool.completed` that drains the last one, so tools + /// running concurrently within a turn are counted once. This matches how + /// the in-process stopwatch brackets `execute_tool_calls`; summing + /// per-call durations would over-count parallel tool use. + #[serde(default, skip_serializing_if = "is_zero_ms")] + pub live_tool_ms: u64, + /// Open tool batch for this stage: when the current batch started, and the + /// `tool_call_id`s that have not yet reported completion. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_batch: Option, #[serde(default)] pub usage: BilledTokenCounts, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -395,6 +416,30 @@ pub struct StageProjection { pub state: StageState, } +/// Serde guard so zero-valued live accumulators stay off the wire. +#[allow( + clippy::trivially_copy_pass_by_ref, + reason = "serde skip_serializing_if predicates receive fields by reference" +)] +fn is_zero_ms(value: &u64) -> bool { + *value == 0 +} + +/// One open tool batch: tool calls dispatched together that have not all +/// reported completion. +/// +/// `open_call_ids` is a set rather than a count because `agent.tool.completed` +/// identifies its call by id, and a projection replaying a truncated or +/// duplicated log must not let a repeated completion drain the batch early. +#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +pub struct StageToolBatchProjection { + /// When the batch opened — the first `agent.tool.started` observed while + /// no other calls were outstanding. + pub started_at: DateTime, + /// Calls dispatched but not yet completed, by `tool_call_id`. + pub open_call_ids: BTreeSet, +} + /// One open inference bracket: a dispatched LLM request that has not yet /// produced a message, error, or interrupt. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] @@ -506,6 +551,9 @@ impl StageProjection { response: None, completion: None, timing: None, + live_inference_ms: 0, + live_tool_ms: 0, + tool_batch: None, usage: BilledTokenCounts::default(), model: None, root_agent_todos: None, @@ -563,6 +611,118 @@ impl StageProjection { self.timing.map(|timing| timing.wall_time_ms) } + /// Live timing breakdown in milliseconds — the active-time twin of + /// [`Self::live_wall_time_ms`]. + /// + /// Once terminal, returns the stored `timing` unchanged: the finalized + /// breakdown comes from the worker's own stopwatch and is authoritative. + /// + /// While in flight, returns an estimate reconstructed from the event log: + /// accumulated closed brackets plus whatever bracket is open right now. + /// The estimate is per-handler, because only agent stages emit brackets at + /// all: + /// + /// - `Agent` — accumulated inference and tool brackets, plus the open + /// inference bracket and open tool batch. + /// - `Prompt` — one inference call spanning the stage, so elapsed time + /// since `started_at` counts as inference. Matches the finalized + /// `active_only(inference, 0)`. + /// - `Command` — the command *is* the work, so elapsed time counts as tool. + /// Matches the finalized `active_only(0, duration_ms)`. + /// - Everything else — zero. Waiting on a human, a timer, a condition, or + /// child branches is wall time, not active time. + /// + /// Active is clamped to wall. A worker killed mid-turn leaves its bracket + /// open forever (see [`StageInferenceProjection`]), and without the clamp + /// that bracket would tick up without bound. The clamp does not need to + /// know the worker died: a stage cannot have been active longer than it + /// has existed. `watchdog.timeout` remains the authority on whether a run + /// is actually stuck. + #[must_use] + pub fn live_timing(&self, now: DateTime) -> StageTiming { + if let Some(timing) = self.timing { + return timing; + } + + let wall_time_ms = self.live_wall_time_ms(now).unwrap_or(0); + let elapsed = |since: DateTime| { + u64::try_from(now.signed_duration_since(since).num_milliseconds().max(0)).unwrap_or(0) + }; + + // `handler` is absent on projections built from events written before + // stage execution identity. Treat those as agent stages, matching + // `StageHandler::from_handler_type`: the accumulators below are only + // ever populated by agent events, so a legacy non-agent stage still + // reads zero rather than being credited work it never did. + let handler = self.handler.unwrap_or(StageHandler::Agent); + let (inference_time_ms, tool_time_ms) = match handler { + StageHandler::Agent => { + let open_inference = self + .inference + .as_ref() + .map_or(0, |inference| elapsed(inference.started_at)); + let open_tool = self + .tool_batch + .as_ref() + .map_or(0, |batch| elapsed(batch.started_at)); + ( + self.live_inference_ms.saturating_add(open_inference), + self.live_tool_ms.saturating_add(open_tool), + ) + } + StageHandler::Prompt => (wall_time_ms, 0), + StageHandler::Command => (0, wall_time_ms), + // Waiting on a human, a timer, a condition, or child branches is + // wall time, not active time. + StageHandler::Human + | StageHandler::Wait + | StageHandler::Conditional + | StageHandler::Parallel + | StageHandler::ParallelFanIn + | StageHandler::StackManagerLoop + | StageHandler::Start + | StageHandler::Exit => (0, 0), + }; + + StageTiming::new(wall_time_ms, inference_time_ms, tool_time_ms).clamped_to_wall() + } + + /// Fold a closed inference bracket into the live accumulator. + pub fn accumulate_inference_ms(&mut self, elapsed_ms: u64) { + self.live_inference_ms = self.live_inference_ms.saturating_add(elapsed_ms); + } + + /// Record a dispatched tool call, opening a batch if none is outstanding. + pub fn open_tool_call(&mut self, tool_call_id: String, started_at: DateTime) { + self.tool_batch + .get_or_insert_with(|| StageToolBatchProjection { + started_at, + open_call_ids: BTreeSet::new(), + }) + .open_call_ids + .insert(tool_call_id); + } + + /// Retire a tool call. Folds the batch into the live accumulator once the + /// last outstanding call reports, so concurrent calls count once. + pub fn close_tool_call(&mut self, tool_call_id: &str, now: DateTime) { + let Some(batch) = self.tool_batch.as_mut() else { + return; + }; + batch.open_call_ids.remove(tool_call_id); + if !batch.open_call_ids.is_empty() { + return; + } + let elapsed = u64::try_from( + now.signed_duration_since(batch.started_at) + .num_milliseconds() + .max(0), + ) + .unwrap_or(0); + self.live_tool_ms = self.live_tool_ms.saturating_add(elapsed); + self.tool_batch = None; + } + /// Begin a new automatic attempt within this stage execution: clear every /// per-attempt field so prior-attempt data does not leak, then record /// `started_at` and `state = Running`. Preserves `first_event_seq` @@ -709,10 +869,13 @@ impl RunProjection { /// terminal conclusion yet. /// /// Run-level wall time ticks from `run.started` to `now`. Active time sums - /// inference and tool timing from stages that have already emitted a - /// terminal stage event. Stage projections do not currently track live - /// inference/tool time while a stage is still running, so active time steps - /// forward when each stage completes while wall time advances continuously. + /// [`StageProjection::live_timing`] across every stage, so an in-flight + /// stage contributes its live estimate rather than nothing — both halves + /// advance continuously. Terminal stages contribute their finalized, + /// authoritative breakdown. + /// + /// Active is not clamped to run wall time here: concurrent branches can + /// legitimately sum past it. The clamp applies per stage. #[must_use] pub fn live_run_timing(&self, now: DateTime) -> Option { let start = self.start.as_ref()?; @@ -720,7 +883,7 @@ impl RunProjection { let active = self .stages .values() - .filter_map(|stage| stage.timing) + .map(|stage| stage.live_timing(now)) .fold(RunTiming::default(), |acc, timing| { acc.saturating_add(&RunTiming::from(timing)) }); @@ -971,3 +1134,286 @@ mod iter_stages_tests { } } } + +#[cfg(test)] +mod live_timing_tests { + use std::collections::HashMap; + use std::num::NonZeroU32; + + use chrono::{DateTime, TimeZone, Utc}; + + use super::{RunProjection, StageToolBatchProjection}; + use crate::{ + Graph, ModelRef, RunId, RunSpec, StageHandler, StageInferenceProjection, StageProjection, + StageState, StageTiming, StartRecord, WorkflowSettings, test_support, + }; + + fn seq(n: u32) -> NonZeroU32 { + NonZeroU32::new(n).unwrap() + } + + fn at(seconds: i64) -> DateTime { + Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap() + } + + fn projection() -> RunProjection { + RunProjection::new( + "Test run".to_string(), + RunSpec { + run_id: RunId::new(), + settings: WorkflowSettings::default(), + graph: Graph::new("test"), + graph_source: None, + workflow_slug: None, + automation: None, + source_directory: None, + labels: HashMap::default(), + provenance: test_support::test_run_provenance(), + manifest_blob: None, + definition_blob: None, + git: None, + fork_source_ref: None, + }, + at(0), + ) + } + + /// In-flight stage that started at `at(0)`. + fn running(handler: StageHandler) -> StageProjection { + let mut stage = StageProjection::new(seq(1)); + stage.handler = Some(handler); + stage.started_at = Some(at(0)); + stage.state = StageState::Running; + stage + } + + fn open_bracket(started_at: DateTime) -> StageInferenceProjection { + StageInferenceProjection { + session_id: "session-1".to_string(), + started_at, + requested_model: ModelRef { + provider: "anthropic".parse().unwrap(), + model_id: "claude-sonnet-5".into(), + speed: None, + }, + first_output_at: None, + first_output_kind: None, + retries: 0, + } + } + + #[test] + fn terminal_stage_returns_stored_timing_unchanged() { + let mut stage = running(StageHandler::Agent); + stage.state = StageState::Succeeded; + stage.timing = Some(StageTiming::new(90_000, 78_000, 7_000)); + // Live accumulators are stale leftovers; the finalized value wins. + stage.live_inference_ms = 5; + stage.live_tool_ms = 5; + + assert_eq!( + stage.live_timing(at(600)), + StageTiming::new(90_000, 78_000, 7_000) + ); + } + + #[test] + fn agent_stage_sums_accumulators_and_open_brackets() { + let mut stage = running(StageHandler::Agent); + stage.live_inference_ms = 30_000; + stage.live_tool_ms = 5_000; + // Inference open for 20s, tools open for 10s, at t=120s. + stage.inference = Some(open_bracket(at(100))); + stage.tool_batch = Some(StageToolBatchProjection { + started_at: at(110), + open_call_ids: ["call-1".to_string()].into_iter().collect(), + }); + + assert_eq!( + stage.live_timing(at(120)), + StageTiming::new(120_000, 50_000, 15_000) + ); + } + + #[test] + fn agent_stage_without_brackets_reports_only_accumulators() { + let mut stage = running(StageHandler::Agent); + stage.live_inference_ms = 30_000; + stage.live_tool_ms = 5_000; + + assert_eq!( + stage.live_timing(at(120)), + StageTiming::new(120_000, 30_000, 5_000) + ); + } + + #[test] + fn prompt_stage_counts_elapsed_as_inference() { + let stage = running(StageHandler::Prompt); + + assert_eq!( + stage.live_timing(at(45)), + StageTiming::new(45_000, 45_000, 0) + ); + } + + #[test] + fn command_stage_counts_elapsed_as_tool() { + let stage = running(StageHandler::Command); + + assert_eq!( + stage.live_timing(at(45)), + StageTiming::new(45_000, 0, 45_000) + ); + } + + #[test] + fn waiting_handlers_report_wall_time_with_zero_active() { + for handler in [ + StageHandler::Human, + StageHandler::Wait, + StageHandler::Conditional, + StageHandler::Parallel, + StageHandler::ParallelFanIn, + StageHandler::StackManagerLoop, + StageHandler::Start, + StageHandler::Exit, + ] { + let stage = running(handler); + let timing = stage.live_timing(at(600)); + + assert_eq!( + timing, + StageTiming::new(600_000, 0, 0), + "{handler} should report wall time only" + ); + } + } + + #[test] + fn open_bracket_from_a_killed_worker_is_clamped_to_wall() { + let mut stage = running(StageHandler::Agent); + // Bracket opened before the stage even started — the pathological + // shape a killed worker leaves behind. Without the clamp this would + // report 700s of inference against 600s of wall. + stage.inference = Some(open_bracket(at(-100))); + + let timing = stage.live_timing(at(600)); + + assert_eq!(timing.wall_time_ms, 600_000); + assert_eq!(timing.active_time_ms, 600_000); + } + + #[test] + fn clamping_preserves_the_inference_tool_split() { + let mut stage = running(StageHandler::Agent); + // 3:1 inference:tool, totalling 200s of active against 100s of wall. + stage.live_inference_ms = 150_000; + stage.live_tool_ms = 50_000; + + let timing = stage.live_timing(at(100)); + + assert_eq!(timing.wall_time_ms, 100_000); + assert_eq!(timing.active_time_ms, 100_000); + assert_eq!(timing.inference_time_ms, 75_000); + assert_eq!(timing.tool_time_ms, 25_000); + } + + #[test] + fn live_run_timing_counts_in_flight_stages_not_just_terminal_ones() { + // The shape that motivated this change: two finished stages and one + // long-running agent stage that had been active nearly the whole run. + let mut projection = projection(); + projection.start = Some(StartRecord { + start_time: at(0), + run_branch: None, + base_sha: None, + }); + + let baseline = projection.stage_entry("baseline", 1, seq(1)); + baseline.handler = Some(StageHandler::Command); + baseline.state = StageState::Succeeded; + baseline.timing = Some(StageTiming::new(42_666, 0, 42_663)); + + let assess = projection.stage_entry("assess", 1, seq(2)); + assess.handler = Some(StageHandler::Agent); + assess.state = StageState::Succeeded; + assess.timing = Some(StageTiming::new(86_025, 78_230, 7_588)); + + let plan = projection.stage_entry("plan", 1, seq(3)); + plan.handler = Some(StageHandler::Agent); + plan.started_at = Some(at(146)); + plan.state = StageState::Running; + plan.live_inference_ms = 700_000; + plan.live_tool_ms = 150_000; + + let timing = projection.live_run_timing(at(1_013)).unwrap(); + + assert_eq!(timing.wall_time_ms, 1_013_000); + assert_eq!(timing.inference_time_ms, 778_230); + assert_eq!(timing.tool_time_ms, 200_251); + // Before this change the in-flight stage contributed nothing and the + // run reported 128,481 ms of active time against 1,013,000 ms of wall. + assert_eq!(timing.active_time_ms, 978_481); + } + + #[test] + fn live_run_timing_may_exceed_run_wall_when_branches_overlap() { + let mut projection = projection(); + projection.start = Some(StartRecord { + start_time: at(0), + run_branch: None, + base_sha: None, + }); + + for (index, node) in ["branch-a", "branch-b", "branch-c"].iter().enumerate() { + let stage = projection.stage_entry(node, 1, seq(u32::try_from(index).unwrap() + 1)); + stage.handler = Some(StageHandler::Agent); + stage.state = StageState::Succeeded; + stage.timing = Some(StageTiming::new(60_000, 60_000, 0)); + } + + let timing = projection.live_run_timing(at(60)).unwrap(); + + assert_eq!(timing.wall_time_ms, 60_000); + assert_eq!( + timing.active_time_ms, 180_000, + "concurrent branches legitimately sum past run wall time" + ); + } +} + +#[cfg(test)] +mod live_timing_legacy_tests { + use chrono::{DateTime, TimeZone, Utc}; + + use crate::{StageProjection, StageState, StageTiming, first_event_seq}; + + fn at(seconds: i64) -> DateTime { + Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap() + } + + #[test] + fn a_legacy_stage_without_a_recorded_handler_uses_its_accumulators() { + let mut stage = StageProjection::new(first_event_seq(1)); + stage.handler = None; + stage.started_at = Some(at(0)); + stage.state = StageState::Running; + stage.live_inference_ms = 30_000; + + assert_eq!( + stage.live_timing(at(120)), + StageTiming::new(120_000, 30_000, 0) + ); + } + + #[test] + fn a_legacy_stage_with_no_accumulators_reports_no_active_time() { + let mut stage = StageProjection::new(first_event_seq(1)); + stage.handler = None; + stage.started_at = Some(at(0)); + stage.state = StageState::Running; + + assert_eq!(stage.live_timing(at(120)), StageTiming::new(120_000, 0, 0)); + } +} diff --git a/lib/foundation/fabro-types/src/timing.rs b/lib/foundation/fabro-types/src/timing.rs index 0d1e555e8..4fe45a36f 100644 --- a/lib/foundation/fabro-types/src/timing.rs +++ b/lib/foundation/fabro-types/src/timing.rs @@ -62,6 +62,32 @@ impl StageTiming { Self::new(0, inference_time_ms, tool_time_ms) } + /// Scale the breakdown down so `active_time_ms` does not exceed + /// `wall_time_ms`, preserving the inference/tool ratio. + /// + /// Only meaningful for live estimates of a single in-flight stage, where + /// an open bracket left behind by a killed worker would otherwise tick up + /// without bound. Finalized timings come from the worker's stopwatch and + /// already satisfy the invariant. + /// + /// Deliberately *not* applied at run level: concurrent branches can + /// legitimately sum past run wall time. + #[must_use] + pub fn clamped_to_wall(&self) -> Self { + if self.active_time_ms <= self.wall_time_ms { + return *self; + } + // Preserve the split rather than truncating one side, so a clamped + // stage still shows where its time went. Widen for the multiply: the + // quotient is bounded by `wall_time_ms` because `active_time_ms` + // exceeds it here, so it always fits back into u64. + let scaled = u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) + / u128::from(self.active_time_ms); + let inference_time_ms = u64::try_from(scaled).unwrap_or(self.wall_time_ms); + let tool_time_ms = self.wall_time_ms.saturating_sub(inference_time_ms); + Self::new(self.wall_time_ms, inference_time_ms, tool_time_ms) + } + /// Sum two timings field-by-field. Used to aggregate visits of one node /// and to accumulate run-level rollups. #[must_use] diff --git a/lib/packages/fabro-api-client/src/.openapi-generator/FILES b/lib/packages/fabro-api-client/src/.openapi-generator/FILES index e3a1a7c0e..4ae265b76 100644 --- a/lib/packages/fabro-api-client/src/.openapi-generator/FILES +++ b/lib/packages/fabro-api-client/src/.openapi-generator/FILES @@ -487,6 +487,7 @@ models/stage-projection.ts models/stage-state.ts models/stage-summary.ts models/stage-timing.ts +models/stage-tool-batch-projection.ts models/start-record.ts models/start-run-request.ts models/steer-run-request.ts diff --git a/lib/packages/fabro-api-client/src/models/index.ts b/lib/packages/fabro-api-client/src/models/index.ts index 5cd29601d..029a2d3ab 100644 --- a/lib/packages/fabro-api-client/src/models/index.ts +++ b/lib/packages/fabro-api-client/src/models/index.ts @@ -457,6 +457,7 @@ export * from './stage-projection'; export * from './stage-state'; export * from './stage-summary'; export * from './stage-timing'; +export * from './stage-tool-batch-projection'; export * from './start-record'; export * from './start-run-request'; export * from './steer-run-request'; diff --git a/lib/packages/fabro-api-client/src/models/run-timing.ts b/lib/packages/fabro-api-client/src/models/run-timing.ts index fb585b0e9..d4fa62262 100644 --- a/lib/packages/fabro-api-client/src/models/run-timing.ts +++ b/lib/packages/fabro-api-client/src/models/run-timing.ts @@ -15,7 +15,7 @@ /** - * Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. + * Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. For a running run, stages still in flight contribute a live estimate rather than nothing, so wall and active both advance continuously. Unlike `StageTiming`, active is not clamped to wall here — concurrent branches can legitimately sum past run wall time. */ export interface RunTiming { 'wall_time_ms': number; diff --git a/lib/packages/fabro-api-client/src/models/stage-projection.ts b/lib/packages/fabro-api-client/src/models/stage-projection.ts index 57fd342ca..8cb88c0ac 100644 --- a/lib/packages/fabro-api-client/src/models/stage-projection.ts +++ b/lib/packages/fabro-api-client/src/models/stage-projection.ts @@ -60,6 +60,9 @@ import type { StageState } from './stage-state'; import type { StageTiming } from './stage-timing'; // May contain unused imports in some cases // @ts-ignore +import type { StageToolBatchProjection } from './stage-tool-batch-projection'; +// May contain unused imports in some cases +// @ts-ignore import type { SubAgentProjection } from './sub-agent-projection'; // May contain unused imports in some cases // @ts-ignore @@ -96,6 +99,15 @@ export interface StageProjection { */ 'started_at'?: string | null; 'timing'?: StageTiming | null; + /** + * Inference time accumulated from closed brackets during the current attempt. Live estimate only — the authoritative value arrives with the terminal event and lands in `timing`. Excludes the currently open bracket, whose span is measured from `inference.started_at`. + */ + 'live_inference_ms'?: number; + /** + * Tool time accumulated from closed tool batches during the current attempt. A batch spans the first dispatched call through the completion that drains the last outstanding one, so tools running concurrently within a turn are counted once. + */ + 'live_tool_ms'?: number; + 'tool_batch'?: StageToolBatchProjection | null; 'usage': BilledTokenCounts; 'model'?: BillingModelRef | null; 'todos'?: TodoListProjection | null; diff --git a/lib/packages/fabro-api-client/src/models/stage-timing.ts b/lib/packages/fabro-api-client/src/models/stage-timing.ts index d9915c409..5bdd92bb4 100644 --- a/lib/packages/fabro-api-client/src/models/stage-timing.ts +++ b/lib/packages/fabro-api-client/src/models/stage-timing.ts @@ -15,7 +15,7 @@ /** - * Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. + * Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. For a terminal stage these come from the worker\'s own stopwatch and are authoritative. For a stage still in flight they are a live estimate reconstructed from the event log, and `active_time_ms` is clamped to `wall_time_ms`. The estimate is replaced by the authoritative breakdown when the stage reaches a terminal event. */ export interface StageTiming { 'wall_time_ms': number; diff --git a/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts new file mode 100644 index 000000000..a6bd7718c --- /dev/null +++ b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts @@ -0,0 +1,29 @@ +/* tslint:disable */ +/* eslint-disable */ +/** + * Fabro Run API + * HTTP API for managing Fabro workflow run executions. + * + * The version of the OpenAPI document: 0.1.0 + * + * + * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). + * https://openapi-generator.tech + * Do not edit the class manually. + */ + + + +/** + * One open tool batch: tool calls dispatched together that have not all reported completion. `open_call_ids` is a set rather than a count so a duplicated completion in a replayed log cannot drain the batch early. + */ +export interface StageToolBatchProjection { + /** + * When the batch opened — the first dispatched call observed while no other calls were outstanding. + */ + 'started_at': string; + /** + * Calls dispatched but not yet completed, by tool call id. + */ + 'open_call_ids': Array; +} From d669a2d55c0b808e2005624dfef05e583b563487 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 15:15:39 -0400 Subject: [PATCH 03/76] fix(agent): correct background-agent notification delivery Three correctness fixes in the Claude 5 background-agent path, plus cleanups from a reuse/quality/efficiency review pass. Fixes: - Background-agent output was run through skill expansion. A child that wrote a bare path ("cleaned up /tmp") failed the whole parent turn with `Unknown skill: /tmp`, and a child whose output happened to name a real skill had its report replaced by that skill's template. Synthesized harness turns now skip expansion; only text the user typed can invoke a skill. - `begin_shutdown` suppressed the pending notification before deciding whether a shutdown would happen. Stopping an agent that had just finished rejected the stop *and* discarded the result the parent was owed. Suppression now happens only once shutdown is committed. - `spawn_inner` registered the notification after publishing the agent in `state.agents`, so a concurrent `shutdown_all` in that window left a pending entry the monitor never completes, and the parent's drain loop would never see the queue as drained. Registration now precedes publication. - `TaskOutput.timeout` was declared `number` but parsed with `as_u64`, so a schema-valid `30000.0` failed at runtime. - Update the fabro-server alias test for the `sonnet` alias moving to Claude Sonnet 5. Cleanups: - The supervisor renders the notification turn; `Session` no longer knows the envelope format. - Replace six near-identical prompt snapshots with a property test over all eight conditional combinations, keeping the default and all-conditionals snapshots for wording. - Collapse `TodoRuntime`'s two mutexes into one. - Read the prompt vocabulary from the registry instead of hardcoding it. - Drop internal vocabulary from the `SendMessage` tool description. Co-Authored-By: Claude Opus 5 (1M context) --- lib/apps/fabro-server/src/server/tests.rs | 2 +- .../fabro-agent/src/profiles/claude5.rs | 2 +- .../fabro-agent/src/profiles/claude5_tools.rs | 6 +- .../fabro-agent/src/profiles/mod.rs | 61 ++++++------- ...sts__claude5_question_prompt_snapshot.snap | 77 ---------------- ...ubagents_and_question_prompt_snapshot.snap | 87 ------------------- ...ts__claude5_subagents_prompt_snapshot.snap | 81 ----------------- ...b_search_and_question_prompt_snapshot.snap | 79 ----------------- ..._search_and_subagents_prompt_snapshot.snap | 83 ------------------ ...s__claude5_web_search_prompt_snapshot.snap | 73 ---------------- lib/components/fabro-agent/src/session.rs | 85 +++++++++++++++--- lib/components/fabro-agent/src/subagent.rs | 83 +++++++++++++++--- .../fabro-agent/src/todo_runtime.rs | 36 ++++---- 13 files changed, 202 insertions(+), 553 deletions(-) delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap delete mode 100644 lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs index 71b4dfebf..958b2e876 100644 --- a/lib/apps/fabro-server/src/server/tests.rs +++ b/lib/apps/fabro-server/src/server/tests.rs @@ -6583,7 +6583,7 @@ async fn test_model_explicit_provider_alias_returns_canonical_model_id_when_unav let response = app.oneshot(req).await.unwrap(); let body = response_json!(response, StatusCode::OK).await; - assert_eq!(body["model_id"], "claude-sonnet-4-6"); + assert_eq!(body["model_id"], "claude-sonnet-5"); assert_eq!(body["provider"], "anthropic"); assert_eq!(body["status"], "skip"); } diff --git a/lib/components/fabro-agent/src/profiles/claude5.rs b/lib/components/fabro-agent/src/profiles/claude5.rs index 6b7826584..d6d2faa6c 100644 --- a/lib/components/fabro-agent/src/profiles/claude5.rs +++ b/lib/components/fabro-agent/src/profiles/claude5.rs @@ -134,7 +134,7 @@ impl AgentProfile for Claude5Profile { skills: &[Skill], ) -> String { let template = EmbeddedPrompt::new("claude5.md.j2", CORE_PROMPT) - .with_vocabulary(ToolVocabulary::Claude5) + .with_vocabulary(self.base.registry.vocabulary()) .with_bool( "has_agent", self.base diff --git a/lib/components/fabro-agent/src/profiles/claude5_tools.rs b/lib/components/fabro-agent/src/profiles/claude5_tools.rs index 20f284c2b..663161bab 100644 --- a/lib/components/fabro-agent/src/profiles/claude5_tools.rs +++ b/lib/components/fabro-agent/src/profiles/claude5_tools.rs @@ -297,7 +297,7 @@ pub(crate) fn make_task_output_tool(supervisor: SubAgentSupervisor) -> Registere "description": "Whether to wait for completion." }, "timeout": { - "type": "number", + "type": "integer", "minimum": 0, "maximum": 600_000, "default": 30000, @@ -402,13 +402,13 @@ pub(crate) fn make_send_message_tool(supervisor: SubAgentSupervisor) -> Register RegisteredTool { definition: definition( NativeTool::SendMessage, - "Send additional instructions to a running child agent by its Fabro agent ID.", + "Send additional instructions to a running background agent by its task ID.", serde_json::json!({ "type": "object", "properties": { "to": { "type": "string", - "description": "The running Fabro agent ID." + "description": "The background agent task ID." }, "message": { "type": "string", diff --git a/lib/components/fabro-agent/src/profiles/mod.rs b/lib/components/fabro-agent/src/profiles/mod.rs index f3ff26aa1..798e39661 100644 --- a/lib/components/fabro-agent/src/profiles/mod.rs +++ b/lib/components/fabro-agent/src/profiles/mod.rs @@ -785,41 +785,42 @@ mod tests { insta::assert_snapshot!(system_prompt(&claude5_profile(false, false, false))); } - #[test] - fn claude5_web_search_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(true, false, false))); - } - - #[test] - fn claude5_subagents_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(false, true, false))); - } - - #[test] - fn claude5_question_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(false, false, true))); - } - - #[test] - fn claude5_web_search_and_subagents_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(true, true, false))); - } - - #[test] - fn claude5_web_search_and_question_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(true, false, true))); - } - - #[test] - fn claude5_subagents_and_question_prompt_snapshot() { - insta::assert_snapshot!(system_prompt(&claude5_profile(false, true, true))); - } - #[test] fn claude5_all_conditionals_prompt_snapshot() { insta::assert_snapshot!(system_prompt(&claude5_profile(true, true, true))); } + /// The two snapshots above pin the wording of every conditional section. + /// This covers the six intermediate combinations, which only need to show + /// that each section appears exactly when its tool is registered -- as + /// snapshots they were six near-identical copies of the same prose, and any + /// edit to the template invalidated all eight at once. + #[test] + fn claude5_prompt_sections_track_registered_tools() { + for web_search in [false, true] { + for subagents in [false, true] { + for question in [false, true] { + let prompt = system_prompt(&claude5_profile(web_search, subagents, question)); + assert_eq!( + prompt.contains("Use `WebSearch`"), + web_search, + "web_search={web_search} subagents={subagents} question={question}" + ); + assert_eq!( + prompt.contains("# Background agents"), + subagents, + "web_search={web_search} subagents={subagents} question={question}" + ); + assert_eq!( + prompt.contains("# Asking the user"), + question, + "web_search={web_search} subagents={subagents} question={question}" + ); + } + } + } + } + #[test] fn gemini_default_prompt_snapshot() { insta::assert_snapshot!(system_prompt(&gemini_profile(false))); diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap deleted file mode 100644 index c1628ad2a..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_question_prompt_snapshot.snap +++ /dev/null @@ -1,77 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(false, false, true))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - - - -# Asking the user - -Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. - -When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap deleted file mode 100644 index 95d827e7a..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_and_question_prompt_snapshot.snap +++ /dev/null @@ -1,87 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(false, true, true))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - -# Background agents - -Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. - -Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. - -Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. - -An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. - - - -# Asking the user - -Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. - -When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap deleted file mode 100644 index eaaaf1cbb..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_subagents_prompt_snapshot.snap +++ /dev/null @@ -1,81 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(false, true, false))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - -# Background agents - -Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. - -Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. - -Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. - -An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. - - - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap deleted file mode 100644 index 041ac14ff..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_question_prompt_snapshot.snap +++ /dev/null @@ -1,79 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(true, false, true))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - -Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - - - -# Asking the user - -Use `AskUserQuestion` only when blocked on a decision that genuinely belongs to the user and cannot be resolved from the request, code, project instructions, or a sensible default. Do not use it for routine permission to continue. - -When recommending an option, put it first and mark it as recommended. The interface supplies a free-form alternative automatically. - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap deleted file mode 100644 index d67ec831e..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_and_subagents_prompt_snapshot.snap +++ /dev/null @@ -1,83 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(true, true, false))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - -Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - -# Background agents - -Use `Agent` for independent, substantial work or to keep broad exploration out of the parent context. Do not delegate a task and duplicate the same work yourself. - -Agents run in the background by default. Launch independent agents together in one response so they run concurrently. Set `run_in_background` to false when their result is an immediate prerequisite and there is no useful parent work to do meanwhile. - -Background completion and failure notifications arrive automatically. Do not poll `TaskOutput` for ordinary progress. Use it only when you deliberately need to block or inspect status. Use `SendMessage` to add instructions to a running agent and `TaskStop` when a running agent is no longer needed. - -An agent's report is evidence, not automatic proof. Inspect relevant changes or run appropriate verification before reporting delegated implementation as complete. Synthesize the useful result for the user; raw agent reports are not a substitute for the final response. - - - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap b/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap deleted file mode 100644 index b92b72e73..000000000 --- a/lib/components/fabro-agent/src/profiles/snapshots/fabro_agent__profiles__tests__claude5_web_search_prompt_snapshot.snap +++ /dev/null @@ -1,73 +0,0 @@ ---- -source: lib/components/fabro-agent/src/profiles/mod.rs -expression: "system_prompt(&claude5_profile(true, false, false))" ---- -You are Claude, a software engineering agent running in Fabro. Use the tools available in this session to complete software engineering work and to answer questions about the codebase. - -When the request is to explain, diagnose, review, or report status, inspect the relevant evidence and return an assessment. Do not modify files or external state unless the request asks for a change. When the request asks you to build or change something, implement the complete change and verify it. - - -Working directory: /home/test -Is git repository: false -Platform: linux -OS version: Linux 6.1.0 - - -# Harness - -- Text outside tool calls is shown to the user as GitHub-flavored Markdown. -- The user may not see your reasoning or raw tool output. Make the final response self-contained. -- Independent tool calls can run in parallel in one response. Run dependent operations sequentially. -- Follow all project and user instructions included in this prompt. -- Reference code with `file_path:line_number` when a precise location helps. - -# Delivering work - -Work on the request actually given. Do not quietly narrow, widen, or transform its scope. Make routine judgment calls yourself. If different interpretations would materially change the result, finish everything independent of that decision and use `AskUserQuestion` when it is available. - -Keep going until the requested outcome is complete. Do not stop at a plan, a list of next steps, or a promise to do work that can be performed with the available tools. If one part is blocked, complete the remaining independent work and report the blocker precisely. - -Use evidence rather than guesses. When an attempt fails, inspect the error and assumptions before making a focused adjustment. Report outcomes faithfully: state which verification ran, what passed or failed, and what was not checked. - -# Working in the codebase - -- Read relevant code before proposing or making changes. -- This workspace tracks reads before writes. Before every `Edit` or `Write` to an existing file, call `Read` on its current contents. Re-read after the file changes before making another edit to it. -- Prefer `Edit` for targeted changes. `Write` creates a file or deliberately replaces its entire contents. -- Make the smallest complete change that satisfies the request. Avoid unrelated refactors, speculative abstractions, and compatibility machinery without a real requirement. -- Match the surrounding code's structure, naming, formatting, and comment density. Add comments only for constraints or reasoning the code cannot make evident. -- Validate at real boundaries such as external input and APIs; do not add defensive branches for states excluded by established invariants. -- Run the most targeted useful verification first, then broaden it when the risk warrants it. Exercise user-facing behavior when the environment supports doing so. - -# Tool use - -Use `Read`, `Edit`, and `Write` instead of shell commands for file inspection and mutation. `Read` handles UTF-8 text and returns numbered lines; it does not render images, PDFs, or notebooks. - -Use `Bash` for searches, git inspection, builds, tests, package managers, and other terminal operations. Prefer `rg` for content search and `rg --files` with `-g` filters for file discovery. Narrow commands so their output remains useful. - -Each `Bash` call is a fresh foreground, non-login shell. Working-directory and environment changes do not persist between calls; keep dependent `cd` or environment setup in the same command. `timeout` is measured in milliseconds. - -Use `TaskCreate`, `TaskUpdate`, `TaskGet`, and `TaskList` when meaningful multi-step work benefits from visible tracking. Keep statuses current as work progresses. Skip task tracking when it adds no value. - -Use `WebFetch` with both a URL and a prompt describing the information to extract. - -Use `WebSearch` when current external information is needed. Its input is a search query; use `WebFetch` to inspect a specific result. - - -Other registered tools may be supplied by MCP servers or the surrounding Fabro workflow. Follow their definitions. - - - - - -# Communicating with the user - -Before the first tool call, state what you are about to do in one concise sentence. While working, give brief updates only at meaningful milestones, such as finding the cause, changing direction, or completing a major phase. - -Do not expose internal deliberation. Write for a teammate catching up: use complete sentences, explain only details that affect conclusions or next actions, and avoid private shorthand. - -Lead the final response with the outcome. A simple question deserves a direct answer rather than unnecessary sections. Be concise by omitting low-value detail, not by compressing useful explanation into fragments. - -# Context management - -Long sessions may be summarized and continued in a new context window. Treat the supplied summary as the continuation of the same work. Do not wrap up early merely because the session is long. diff --git a/lib/components/fabro-agent/src/session.rs b/lib/components/fabro-agent/src/session.rs index f04e99257..c72d33683 100644 --- a/lib/components/fabro-agent/src/session.rs +++ b/lib/components/fabro-agent/src/session.rs @@ -47,10 +47,7 @@ use crate::skills::{ ExpandedInput, Skill, default_skill_dirs, discover_skills, expand_skill, make_use_skill_tool_for_vocabulary, }; -use crate::subagent::{ - SubAgentCallbackEvent, SubAgentEventCallback, SubAgentSupervisor, - format_parent_notification_batch, -}; +use crate::subagent::{SubAgentCallbackEvent, SubAgentEventCallback, SubAgentSupervisor}; use crate::tool_execution::execute_tool_calls; use crate::tool_permissions::canonical_tool_name; use crate::tool_registry::ToolDefinitionWithSource; @@ -371,6 +368,18 @@ struct BuiltRequest { context_window: StageContextWindowProjection, } +/// Whether an input's `/name` tokens should be treated as skill references. +/// +/// Only text the user actually typed can invoke a skill. Harness-synthesized +/// input carries whatever a child agent wrote, where `/tmp` is a path rather +/// than an invocation: expanding it would either fail the parent turn on an +/// unknown name or splice a skill template in place of the envelope. +#[derive(Clone, Copy, PartialEq, Eq)] +enum SkillExpansion { + Apply, + Skip, +} + pub struct Session { id: String, /// Root agent session ID for this session's agent tree. A root session @@ -1328,6 +1337,7 @@ impl Session { let mut result = self .run_single_input( input, + SkillExpansion::Apply, &agent_tool_runtime, &mut timing, &mut usage, @@ -1343,15 +1353,13 @@ impl Session { .expect("followup queue lock poisoned") .pop_front(); let next_input = if let Some(followup) = followup { - Some(followup) + Some((followup, SkillExpansion::Apply)) } else if let Some(supervisor) = self.subagent_supervisor.clone() { match supervisor - .next_parent_notification_batch(&self.cancel_token) + .next_parent_notification_turn(&self.cancel_token) .await { - Ok(Some(notifications)) => { - Some(format_parent_notification_batch(¬ifications)) - } + Ok(Some(turn)) => Some((turn, SkillExpansion::Skip)), Ok(None) => None, Err(Error::Interrupted(InterruptReason::Cancelled)) => { result = Err(self.interrupted_error()); @@ -1365,10 +1373,13 @@ impl Session { } else { None }; - let Some(next_input) = next_input else { break }; + let Some((next_input, skill_expansion)) = next_input else { + break; + }; result = self .run_single_input( &next_input, + skill_expansion, &agent_tool_runtime, &mut timing, &mut usage, @@ -1406,6 +1417,7 @@ impl Session { async fn run_single_input( &mut self, input: &str, + skill_expansion: SkillExpansion, agent_tool_runtime: &AgentToolRuntime, timing: &mut SessionInputTiming, usage_accumulator: &mut TokenCounts, @@ -1420,7 +1432,7 @@ impl Session { self.transition(SessionState::Thinking); // Expand skill references in input - let expanded = if self.skills.is_empty() { + let expanded = if self.skills.is_empty() || skill_expansion == SkillExpansion::Skip { ExpandedInput { text: input.to_string(), skill_name: None, @@ -3133,6 +3145,57 @@ mod tests { supervisor.shutdown_all().await; } + #[tokio::test] + async fn background_agent_output_is_not_parsed_for_skill_references() { + let supervisor = SubAgentSupervisor::new(3); + let child = make_session(vec![text_response("Cleaned up /tmp and exited")]).await; + let child_id = supervisor + .spawn_with_parent_notification( + child, + "clean up".to_string(), + "Clean scratch files".to_string(), + 0, + ) + .unwrap(); + supervisor + .wait_with_cancel(&child_id, &CancellationToken::new()) + .await + .unwrap(); + + let provider = Arc::new(ScriptedStreamProvider::new(vec![ + ScriptedStreamCall::Response(Box::new(text_response("Delegated"))), + ScriptedStreamCall::Response(Box::new(text_response("Acknowledged"))), + ])); + let mut parent = + make_session_with_provider_and_manager(provider, Some(supervisor.clone())).await; + parent.skills = vec![Skill { + name: "commit".to_string(), + description: "Make a commit".to_string(), + template: "Review changes and commit.".to_string(), + }]; + + // A child that mentions a bare path must not fail the parent turn on + // `Unknown skill: /tmp`, nor have its report replaced by a skill body. + let output = parent + .process_input_with_output("Delegate the cleanup") + .await + .unwrap(); + + assert_eq!(output.as_deref(), Some("Acknowledged")); + let turns = parent.history().turns(); + let Message::User { + content: notification, + .. + } = &turns[2] + else { + panic!("third turn should deliver the background result"); + }; + assert!(notification.contains("Cleaned up /tmp and exited")); + assert!(!notification.contains("Review changes and commit.")); + + supervisor.shutdown_all().await; + } + #[tokio::test] async fn events_emitted() { let mut session = make_session(vec![text_response("Hello")]).await; diff --git a/lib/components/fabro-agent/src/subagent.rs b/lib/components/fabro-agent/src/subagent.rs index 0d91d58e8..f4d6c1b84 100644 --- a/lib/components/fabro-agent/src/subagent.rs +++ b/lib/components/fabro-agent/src/subagent.rs @@ -42,9 +42,7 @@ pub(crate) struct SubAgentParentNotification { pub result: Result, } -pub(crate) fn format_parent_notification_batch( - notifications: &[SubAgentParentNotification], -) -> String { +fn format_parent_notification_batch(notifications: &[SubAgentParentNotification]) -> String { notifications .iter() .map(|notification| { @@ -481,6 +479,16 @@ impl SubAgentSupervisor { child_depth, ); + // Register before the agent becomes discoverable in `state.agents`. + // Once it is, a concurrent `shutdown_all` can suppress and close it; + // registering afterwards would leave a pending entry that the monitor + // never completes (it early-returns for a non-Running agent), and + // `next_batch` would then never report the queue as drained. Nothing + // can complete this registration before `start_tx.send(())` below. + if let Some(description) = parent_notification_description { + self.parent_notifications + .register(agent_id.clone(), description); + } { let mut state = self.state.lock().expect("subagent state lock poisoned"); state.agents.insert(agent_id.clone(), SubAgent { @@ -496,10 +504,6 @@ impl SubAgentSupervisor { depth: child_depth, }); } - if let Some(description) = parent_notification_description { - self.parent_notifications - .register(agent_id.clone(), description); - } self.emit_event(AgentEvent::SubAgentSpawned { agent_id: agent_id.clone(), @@ -591,7 +595,23 @@ impl SubAgentSupervisor { } /// Wait until all currently-ready background results can be delivered in - /// one parent turn, or return `None` once no notifiable agents remain. + /// one parent turn, rendered as the text of that turn. Returns `None` once + /// no notifiable agents remain. + /// + /// The envelope format is the supervisor's concern, so callers receive a + /// finished turn rather than the notifications behind it. + pub(crate) async fn next_parent_notification_turn( + &self, + cancel: &CancellationToken, + ) -> Result, Error> { + Ok(self + .next_parent_notification_batch(cancel) + .await? + .map(|notifications| format_parent_notification_batch(¬ifications))) + } + + /// The notifications behind [`Self::next_parent_notification_turn`], for + /// tests that assert on delivery semantics rather than on the rendering. pub(crate) async fn next_parent_notification_batch( &self, cancel: &CancellationToken, @@ -606,7 +626,6 @@ impl SubAgentSupervisor { } fn begin_shutdown(&self, agent_id: &str, strict: bool) -> Result { - self.parent_notifications.suppress(agent_id); let mut state = self.state.lock().expect("subagent state lock poisoned"); let agent = state.agents.get_mut(agent_id).ok_or_else(|| { Error::InvalidState(format!( @@ -752,7 +771,12 @@ impl SubAgentSupervisor { } async fn ensure_closed(&self, agent_id: &str) -> Result<(), Error> { - let cleanup_done = match self.begin_shutdown(agent_id, false)? { + let disposition = self.begin_shutdown(agent_id, false)?; + // Only once shutdown is committed. Suppressing before `begin_shutdown` + // would also discard the result of an agent that had already finished, + // which rejects the shutdown but had a delivery pending. + self.parent_notifications.suppress(agent_id); + let cleanup_done = match disposition { ShutdownDisposition::Lead(work) => self.spawn_shutdown(work), ShutdownDisposition::Follow(cleanup_done) => cleanup_done, ShutdownDisposition::Done => return Ok(()), @@ -763,7 +787,9 @@ impl SubAgentSupervisor { /// Strict user-facing close: only a currently running child may be closed. pub async fn close_agent(&self, agent_id: &str) -> Result<(), Error> { - let cleanup_done = match self.begin_shutdown(agent_id, true)? { + let disposition = self.begin_shutdown(agent_id, true)?; + self.parent_notifications.suppress(agent_id); + let cleanup_done = match disposition { ShutdownDisposition::Lead(work) => self.spawn_shutdown(work), ShutdownDisposition::Follow(_) | ShutdownDisposition::Done => { return Err(Error::InvalidState(format!( @@ -1109,6 +1135,41 @@ mod tests { ); } + #[tokio::test] + async fn rejected_stop_of_a_finished_agent_keeps_its_notification() { + let supervisor = SubAgentSupervisor::new(3); + let child = make_session(vec![text_response("child result")]).await; + let agent_id = supervisor + .spawn_with_parent_notification( + child, + "task".to_string(), + "Inspect the module".to_string(), + 0, + ) + .unwrap(); + + // Finish the child so its result is queued for automatic delivery. + supervisor + .wait_with_cancel(&agent_id, &CancellationToken::new()) + .await + .unwrap(); + + // Stopping a finished agent is rejected... + let error = supervisor.close_agent(&agent_id).await.unwrap_err(); + assert!(matches!(error, Error::InvalidState(_)), "{error:?}"); + + // ...so it must not have discarded the result the parent is owed. + let batch = supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .expect("a rejected stop must leave the pending result deliverable"); + assert_eq!(batch.len(), 1); + assert_eq!(batch[0].agent_id, agent_id); + + supervisor.shutdown_all().await; + } + #[tokio::test] async fn spawn_creates_agent_and_returns_id() { let manager = SubAgentSupervisor::new(3); diff --git a/lib/components/fabro-agent/src/todo_runtime.rs b/lib/components/fabro-agent/src/todo_runtime.rs index 0d1d79e11..dbf2791f5 100644 --- a/lib/components/fabro-agent/src/todo_runtime.rs +++ b/lib/components/fabro-agent/src/todo_runtime.rs @@ -17,20 +17,26 @@ use fabro_types::{ use crate::tool_registry::ToolContext; use crate::types::AgentEvent; +/// Projections and their ID counters, behind one lock so a list and its +/// counter can never be observed out of step. +#[derive(Debug, Default)] +struct TodoRuntimeState { + lists: BTreeMap, + task_counters: BTreeMap, +} + /// Shared, thread-safe todo projection. Wrap it in `Arc` and clone the /// `Arc` into each tool closure that needs it. #[derive(Debug, Default)] pub struct TodoRuntime { - lists: Mutex>, - task_counters: Mutex>, + state: Mutex, } impl TodoRuntime { #[must_use] pub fn new() -> Self { Self { - lists: Mutex::new(BTreeMap::new()), - task_counters: Mutex::new(BTreeMap::new()), + state: Mutex::new(TodoRuntimeState::default()), } } @@ -39,11 +45,8 @@ impl TodoRuntime { /// Keeping the counter beside the projection lets root and child profiles /// safely create tasks in the same shared list. pub(crate) fn next_task_id(&self, list_id: &str) -> u64 { - let mut counters = self - .task_counters - .lock() - .expect("task counter lock poisoned"); - let counter = counters.entry(list_id.to_string()).or_default(); + let mut guard = self.state.lock().expect("todo runtime lock poisoned"); + let counter = guard.task_counters.entry(list_id.to_string()).or_default(); *counter = counter.saturating_add(1); *counter } @@ -52,8 +55,8 @@ impl TodoRuntime { /// list-style tools that need a stable view. #[must_use] pub fn snapshot(&self, list_id: &str) -> Option { - let guard = self.lists.lock().expect("todo runtime lock poisoned"); - guard.get(list_id).cloned() + let guard = self.state.lock().expect("todo runtime lock poisoned"); + guard.lists.get(list_id).cloned() } /// Insert (or replace) a todo and emit `todo.created`. @@ -79,8 +82,9 @@ impl TodoRuntime { metadata: todo.metadata.clone(), }; { - let mut guard = self.lists.lock().expect("todo runtime lock poisoned"); + let mut guard = self.state.lock().expect("todo runtime lock poisoned"); guard + .lists .entry(list_id) .or_insert_with(|| TodoListProjection::new(kind, props.list_id.clone())) .upsert(todo); @@ -97,8 +101,8 @@ impl TodoRuntime { } let applied = { - let mut guard = self.lists.lock().expect("todo runtime lock poisoned"); - let Some(list) = guard.get_mut(&props.list_id) else { + let mut guard = self.state.lock().expect("todo runtime lock poisoned"); + let Some(list) = guard.lists.get_mut(&props.list_id) else { return false; }; list.apply_patch(&props.todo_id, &TodoPatch::from_props(&props)) @@ -119,8 +123,8 @@ impl TodoRuntime { todo_id: String, ) -> bool { let removed = { - let mut guard = self.lists.lock().expect("todo runtime lock poisoned"); - let Some(list) = guard.get_mut(&list_id) else { + let mut guard = self.state.lock().expect("todo runtime lock poisoned"); + let Some(list) = guard.lists.get_mut(&list_id) else { return false; }; list.remove(&todo_id) From dd9f75fb05d55b29b5f2db10d8e0153acc0e95eb Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 23:43:43 -0400 Subject: [PATCH 04/76] fix(timing): harden live active projections --- apps/fabro-web/app/lib/queries.ts | 12 +- apps/fabro-web/app/lib/query-keys.test.ts | 15 +- apps/fabro-web/app/lib/run-events.test.tsx | 20 ++ apps/fabro-web/app/lib/run-events.ts | 29 ++- apps/fabro-web/app/routes/run-detail.tsx | 6 +- ...05-21-wall-and-active-time-metrics-plan.md | 8 +- docs/public/api-reference/fabro-api.yaml | 9 + .../fabro-server/src/server/handler/runs.rs | 30 ++- lib/apps/fabro-server/src/server/tests.rs | 89 +++++++++ lib/components/fabro-store/src/run_state.rs | 182 +++++++++++++++--- lib/foundation/fabro-api/build.rs | 5 + lib/foundation/fabro-api/src/lib.rs | 8 +- .../tests/stage_projection_round_trip.rs | 31 ++- lib/foundation/fabro-types/src/lib.rs | 2 +- .../fabro-types/src/run_projection.rs | 109 ++++++++--- lib/foundation/fabro-types/src/timing.rs | 48 ++++- .../src/models/stage-tool-batch-projection.ts | 6 +- 17 files changed, 523 insertions(+), 86 deletions(-) diff --git a/apps/fabro-web/app/lib/queries.ts b/apps/fabro-web/app/lib/queries.ts index 3c8eb8a1d..aa7328b5e 100644 --- a/apps/fabro-web/app/lib/queries.ts +++ b/apps/fabro-web/app/lib/queries.ts @@ -73,6 +73,7 @@ import { type RunFileSelection, type RunGraphDirection, } from "./query-keys"; +import { isTerminalRunStatus } from "./run-actions"; const immutableOptions: SWRConfiguration = { revalidateIfStale: false, @@ -179,10 +180,19 @@ export function useRunsPage(opts: RunsPageOptions = {}, enabled = true) { ); } -export function useRun(id: string | undefined) { +export function useRun(id: string | undefined, refreshInterval?: number) { return useSWR( id ? queryKeys.runs.detail(id) : null, () => apiNullableData(() => runsApi.retrieveRun(id!)), + refreshInterval + ? { + refreshInterval: (run) => + run?.timestamps.started_at && + !isTerminalRunStatus(run.lifecycle.status.kind) + ? refreshInterval + : 0, + } + : undefined, ); } diff --git a/apps/fabro-web/app/lib/query-keys.test.ts b/apps/fabro-web/app/lib/query-keys.test.ts index ac441b998..52a5ccef3 100644 --- a/apps/fabro-web/app/lib/query-keys.test.ts +++ b/apps/fabro-web/app/lib/query-keys.test.ts @@ -87,8 +87,6 @@ describe("queryKeys", () => { test("agent activity events invalidate per-stage resources", () => { for (const event of [ "stage.prompt", - "agent.tool.started", - "agent.tool.completed", "command.started", "command.completed", ]) { @@ -97,8 +95,19 @@ describe("queryKeys", () => { queryKeys.runs.stageContextWindow("run-1", "stage-1"), ]); } + for (const event of ["agent.tool.started", "agent.tool.completed"]) { + expect(queryKeysForRunEvent("run-1", event, "stage-1")).toEqual([ + queryKeys.runs.detail("run-1"), + queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), + queryKeys.runs.stageEvents("run-1", "stage-1"), + queryKeys.runs.stageContextWindow("run-1", "stage-1"), + ]); + } expect(queryKeysForRunEvent("run-1", "agent.message", "stage-1")).toEqual([ + queryKeys.runs.detail("run-1"), queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), queryKeys.runs.stageEvents("run-1", "stage-1"), queryKeys.runs.stageContextWindow("run-1", "stage-1"), ]); @@ -106,7 +115,9 @@ describe("queryKeys", () => { test("agent message without a node_id still invalidates projected state", () => { expect(queryKeysForRunEvent("run-1", "agent.message")).toEqual([ + queryKeys.runs.detail("run-1"), queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), ]); }); }); diff --git a/apps/fabro-web/app/lib/run-events.test.tsx b/apps/fabro-web/app/lib/run-events.test.tsx index e984be74d..4fce18b4d 100644 --- a/apps/fabro-web/app/lib/run-events.test.tsx +++ b/apps/fabro-web/app/lib/run-events.test.tsx @@ -85,6 +85,8 @@ describe("queryKeysForRunEvent", () => { test("interrupt settlement invalidates projected control state and stage activity", () => { expect(queryKeysForRunEvent("run-1", "agent.round.interrupted", "nap@1")).toEqual([ + queryKeys.runs.detail("run-1"), + queryKeys.runs.billing("run-1"), queryKeys.runs.state("run-1"), queryKeys.runs.events("run-1", 1000), queryKeys.runs.stageEvents("run-1", "nap@1"), @@ -134,22 +136,40 @@ describe("queryKeysForRunEvent", () => { "agent.error", ]) { expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([ + queryKeys.runs.detail("run-1"), queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), queryKeys.runs.stageEvents("run-1", "code@1"), ]); } expect( queryKeysForRunEvent("run-1", "agent.message", "code@1"), ).toEqual([ + queryKeys.runs.detail("run-1"), queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), queryKeys.runs.stageEvents("run-1", "code@1"), queryKeys.runs.stageContextWindow("run-1", "code@1"), ]); expect(queryKeysForRunEvent("run-1", "agent.session.ended")).toEqual([ + queryKeys.runs.detail("run-1"), queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), ]); }); + test("tool timing events invalidate live summaries and stage resources", () => { + for (const event of ["agent.tool.started", "agent.tool.completed"]) { + expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([ + queryKeys.runs.detail("run-1"), + queryKeys.runs.state("run-1"), + queryKeys.runs.billing("run-1"), + queryKeys.runs.stageEvents("run-1", "code@1"), + queryKeys.runs.stageContextWindow("run-1", "code@1"), + ]); + } + }); + test("watchdog timeout refreshes the stage events for that stage", () => { expect( queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"), diff --git a/apps/fabro-web/app/lib/run-events.ts b/apps/fabro-web/app/lib/run-events.ts index d0d4f4b4d..523774e18 100644 --- a/apps/fabro-web/app/lib/run-events.ts +++ b/apps/fabro-web/app/lib/run-events.ts @@ -118,6 +118,10 @@ const INFERENCE_EVENTS = new Set([ "agent.error", "agent.session.ended", ]); +const TOOL_TIMING_EVENTS = new Set([ + "agent.tool.started", + "agent.tool.completed", +]); // Todo / task mutation events refresh `getRunState` consumers (so per-stage // todo projections update live) and the run events list. const TODO_EVENTS = new Set([ @@ -184,6 +188,12 @@ export function queryKeysForRunEvent( if (AGENT_CONTROL_STATE_EVENTS.has(event)) { keys.unshift(queryKeys.runs.state(runId)); } + if (event === "agent.round.interrupted") { + keys.unshift( + queryKeys.runs.detail(runId), + queryKeys.runs.billing(runId), + ); + } if (stageId) { keys.push(queryKeys.runs.stageEvents(runId, stageId)); keys.push(queryKeys.runs.stageContextWindow(runId, stageId)); @@ -192,7 +202,11 @@ export function queryKeysForRunEvent( } if (INFERENCE_EVENTS.has(event)) { - const keys: Key[] = [queryKeys.runs.state(runId)]; + const keys: Key[] = [ + queryKeys.runs.detail(runId), + queryKeys.runs.state(runId), + queryKeys.runs.billing(runId), + ]; if (stageId) { keys.push(queryKeys.runs.stageEvents(runId, stageId)); if (event === "agent.message") { @@ -202,6 +216,19 @@ export function queryKeysForRunEvent( return keys; } + if (TOOL_TIMING_EVENTS.has(event)) { + const keys: Key[] = [ + queryKeys.runs.detail(runId), + queryKeys.runs.state(runId), + queryKeys.runs.billing(runId), + ]; + if (stageId) { + keys.push(queryKeys.runs.stageEvents(runId, stageId)); + keys.push(queryKeys.runs.stageContextWindow(runId, stageId)); + } + return keys; + } + if (event === "watchdog.timeout") { return stageId ? [queryKeys.runs.stageEvents(runId, stageId)] : []; } diff --git a/apps/fabro-web/app/routes/run-detail.tsx b/apps/fabro-web/app/routes/run-detail.tsx index 97ce00702..ad2d78986 100644 --- a/apps/fabro-web/app/routes/run-detail.tsx +++ b/apps/fabro-web/app/routes/run-detail.tsx @@ -69,6 +69,8 @@ import { export const handle = { hideHeader: true }; +const RUN_TIMING_REFRESH_INTERVAL_MS = 30_000; + type LifecycleTrigger = () => Promise; export function meta({ data }: any) { @@ -78,7 +80,7 @@ export function meta({ data }: any) { export default function RunDetail({ params }: { params: { id: string } }) { const demoMode = useDemoMode(); - const runQuery = useRun(params.id); + const runQuery = useRun(params.id, RUN_TIMING_REFRESH_INTERVAL_MS); const runStateQuery = useRunState(params.id); const summary = runQuery.data; const run = summary ? buildRunDetailRun(summary) : null; @@ -116,7 +118,7 @@ export default function RunDetail({ params }: { params: { id: string } }) { childrenCount, }); const steerBarRef = useRef(null); - const now = useTickingNow(30_000); + const now = useTickingNow(RUN_TIMING_REFRESH_INTERVAL_MS); const { fullHeight, hideSteerBar } = childRouteLayoutFlags(matches); useRunEvents(params.id); diff --git a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md index 3683e5879..c0980ed01 100644 --- a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md +++ b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md @@ -155,8 +155,9 @@ git diff --check - Inference time is Fabro-observed LLM request/stream elapsed time, not provider-reported model-only compute time. -- LLM retry backoff, queueing outside a request/stream, human waits, steering - waits, and scheduler gaps are wall time but not active time. +- Queueing outside a request/stream, human waits, steering waits, and scheduler + gaps are wall time but not active time. Retry delay inside an open LLM request + bracket follows the executor stopwatch and counts as inference time. - ~~Active timing is finalized-event based in v1; live active-time ticking can be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became necessary: a run parked in one long agent stage reported ~12% of its wall time @@ -164,7 +165,6 @@ git diff --check accumulate inference and tool brackets from the event log and expose `StageProjection::live_timing(now)`, the active-time twin of `live_wall_time_ms`. Finalized values remain authoritative and still replace - the live estimate at terminal events. See - `.ai/plans/live-active-time-accumulation.md`. + the live estimate at terminal events. Implemented in PR #647. - No compatibility layer is required for existing API clients or stored run event data. diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml index 568aa30e5..f2460a69b 100644 --- a/docs/public/api-reference/fabro-api.yaml +++ b/docs/public/api-reference/fabro-api.yaml @@ -10772,9 +10772,16 @@ components: duplicated completion in a replayed log cannot drain the batch early. type: object required: + - session_id - started_at - open_call_ids properties: + session_id: + type: string + description: > + Root agent session that dispatched the batch. Transitions are + gated on it so delayed events from a replaced session cannot + mutate the current batch. started_at: type: string format: date-time @@ -10783,6 +10790,8 @@ components: other calls were outstanding. open_call_ids: type: array + minItems: 1 + uniqueItems: true items: type: string description: Calls dispatched but not yet completed, by tool call id. diff --git a/lib/apps/fabro-server/src/server/handler/runs.rs b/lib/apps/fabro-server/src/server/handler/runs.rs index 0e2c8cbcb..4b9acdf1c 100644 --- a/lib/apps/fabro-server/src/server/handler/runs.rs +++ b/lib/apps/fabro-server/src/server/handler/runs.rs @@ -11,7 +11,7 @@ use axum_extra::extract::Query as ExtraQuery; use base64::Engine as _; use base64::engine::general_purpose::STANDARD as BASE64_STANDARD; use bytes::Bytes; -use chrono::Utc; +use chrono::{DateTime, Utc}; use fabro_api::types::{ BoardColumn, RunManifest, SubmitAnswerRequest, UpdateRunParentRequest, UpdateRunRequest, }; @@ -23,7 +23,7 @@ use fabro_store::{ }; use fabro_types::settings::ResolveError; use fabro_types::{ - AutomationRef, Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance, + AutomationRef, Principal, Run, RunClientProvenance, RunId, RunProvenance, RunServerProvenance, RunStatusKind, StageContextWindow, StageContextWindowStaleness, StageContextWindowUnavailableReason, StageHandler, StageModelUsage, StageProjection, SystemActorKind, WorkflowSettings, parse_blob_ref, @@ -298,7 +298,7 @@ async fn validate_parent_link( } async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response { - match state.stores.run_summaries.get(run_id, Utc::now()).await { + match run_summary_at(state, run_id, Utc::now()).await { Ok(Some(summary)) => ( StatusCode::OK, Json(state.decorate_run_summary(summary).await), @@ -311,6 +311,28 @@ async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response { } } +/// Read the durable summary and overlay its timing from the live projection. +/// +/// The SQLite read model stores active timing as of the most recent event. +/// An open inference or tool bracket keeps accruing between events, so detail +/// reads need the projection's current estimate while the run is non-terminal. +async fn run_summary_at( + state: &AppState, + run_id: &RunId, + now: DateTime, +) -> fabro_store::Result> { + let Some(mut summary) = state.stores.run_summaries.get(run_id, now).await? else { + return Ok(None); + }; + if summary.timestamps.completed_at.is_none() { + let cached = state.stores.runs.get_cached_run(run_id).await?; + if let Some(timing) = cached.and_then(|cached| cached.projection.live_run_timing(now)) { + summary.timing = Some(timing); + } + } + Ok(Some(summary)) +} + async fn list_runs( _auth: RequiredRunManagementActor, State(state): State>, @@ -934,7 +956,7 @@ async fn get_run_status( RequireRunManagementTarget(id, _actor): RequireRunManagementTarget, State(state): State>, ) -> Response { - match state.stores.run_summaries.get(&id, Utc::now()).await { + match run_summary_at(&state, &id, Utc::now()).await { Ok(Some(run)) => { (StatusCode::OK, Json(state.decorate_run_summary(run).await)).into_response() } diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs index 71b4dfebf..790089ab1 100644 --- a/lib/apps/fabro-server/src/server/tests.rs +++ b/lib/apps/fabro-server/src/server/tests.rs @@ -5503,6 +5503,50 @@ async fn list_run_stages_exposes_execution_identity_for_resumed_stage() { assert_eq!(second["resumed_from_stage_id"], "work@1"); } +#[tokio::test] +async fn run_billing_includes_live_stage_timing_in_rows_and_totals() { + let state = test_app_state_with_isolated_storage(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = RunId::new(); + create_durable_run_with_events(&state, run_id, &[ + workflow_event::Event::RunSubmitted { + definition_blob: None, + }, + workflow_event::Event::RunStarting, + workflow_event::Event::RunRunning, + workflow_run_started_event(run_id), + ]) + .await; + append_scoped_stage_event( + &state, + run_id, + "work", + 1, + &stage_started_event("work", "command"), + ) + .await; + + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + let response = app + .oneshot( + Request::builder() + .method("GET") + .uri(api(&format!("/runs/{run_id}/billing"))) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + let body = response_json!(response, StatusCode::OK).await; + let stages = body["stages"].as_array().unwrap(); + + assert_eq!(stages.len(), 1); + let row_timing = &stages[0]["timing"]; + assert!(row_timing["active_time_ms"].as_u64().unwrap() > 0); + assert_eq!(row_timing["tool_time_ms"], row_timing["active_time_ms"]); + assert_eq!(&body["totals"]["timing"], row_timing); +} + /// `checkpoint.completed_nodes` records every visit, so a looped node appears /// once per re-entry. Billing must dedup so a retried node renders as one row /// and `runtime_secs` is summed across all visits exactly once. @@ -7800,6 +7844,51 @@ async fn get_run_status_returns_status() { assert!(body["labels"].is_object()); } +#[tokio::test] +async fn get_run_status_advances_live_active_timing_between_events() { + let state = test_app_state_with_isolated_storage(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = RunId::new(); + create_durable_run_with_events(&state, run_id, &[ + workflow_event::Event::RunSubmitted { + definition_blob: None, + }, + workflow_event::Event::RunStarting, + workflow_event::Event::RunRunning, + workflow_run_started_event(run_id), + ]) + .await; + append_scoped_stage_event( + &state, + run_id, + "work", + 1, + &stage_started_event("work", "command"), + ) + .await; + + // The SQLite summary stores timing at the StageStarted event. A later + // detail read must overlay the in-flight command's active time from the + // projection even though no newer event has arrived. + tokio::time::sleep(std::time::Duration::from_millis(20)).await; + let response = app + .oneshot( + Request::builder() + .method("GET") + .uri(api(&format!("/runs/{run_id}"))) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + let body = response_json!(response, StatusCode::OK).await; + let timing = &body["timing"]; + + assert!(timing["active_time_ms"].as_u64().unwrap() > 0); + assert_eq!(timing["tool_time_ms"], timing["active_time_ms"]); + assert!(timing["wall_time_ms"].as_u64().unwrap() >= timing["active_time_ms"].as_u64().unwrap()); +} + #[tokio::test] async fn get_run_status_not_found() { let app = test_app_with(); diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs index df05f84b3..14e86124a 100644 --- a/lib/components/fabro-store/src/run_state.rs +++ b/lib/components/fabro-store/src/run_state.rs @@ -19,6 +19,7 @@ use fabro_types::{ SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection, StageModelUsage, StageOutcome, StageProjection, StageState, StartRecord, SubAgentProjection, SubAgentStatus, TodoListKind, TodoListProjection, TodoProjection, WorkflowRef, first_event_seq, + timing, }; use fabro_util::error::render_compact_with_causes; @@ -405,7 +406,7 @@ impl RunProjectionReducer for RunProjection { }; stage.response = response; stage.completion = Some(completion); - stage.timing = Some(props.timing); + stage.set_authoritative_timing(props.timing); if let Some(billing) = &props.billing { stage.usage.replace_with_billed_usage(billing); stage.model = Some(billing.model().clone()); @@ -428,7 +429,7 @@ impl RunProjectionReducer for RunProjection { failure_reason, timestamp: ts, }); - stage.timing = Some(props.timing); + stage.set_authoritative_timing(props.timing); if let Some(billing) = &props.billing { stage.usage.replace_with_billed_usage(billing); stage.model = Some(billing.model().clone()); @@ -481,7 +482,7 @@ impl RunProjectionReducer for RunProjection { close_inference_bracket(self, stored, props.visit, event.seq, ts); } EventBody::AgentSessionEnded(_) => { - close_inference_brackets_for_session(self, stored, ts); + close_active_brackets_for_session(self, stored, ts); } EventBody::AgentSessionActivated(props) => { let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) @@ -746,7 +747,11 @@ impl RunProjectionReducer for RunProjection { }); } EventBody::AgentToolStarted(props) => { - let is_root_session = stored.parent_session_id.is_none(); + let root_session_id = if stored.parent_session_id.is_none() { + stored.session_id.clone() + } else { + None + }; let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) else { return Ok(()); @@ -770,19 +775,22 @@ impl RunProjectionReducer for RunProjection { // A subagent's tools run inside the root session's tool call, // so the root batch already covers them. Timing them again // would double-count that span. - if is_root_session { - stage.open_tool_call(props.tool_call_id.clone(), ts); + if let Some(session_id) = root_session_id { + stage.open_tool_call(session_id, props.tool_call_id.clone(), ts); } } EventBody::AgentToolCompleted(props) => { if stored.parent_session_id.is_some() { return Ok(()); } + let Some(session_id) = stored.session_id.as_deref() else { + return Ok(()); + }; let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) else { return Ok(()); }; - stage.close_tool_call(&props.tool_call_id, ts); + stage.close_tool_call(session_id, &props.tool_call_id, ts); } _ => {} } @@ -1123,16 +1131,10 @@ fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime) { let Some(inference) = stage.inference.take() else { return; }; - stage.accumulate_inference_ms(elapsed_ms(inference.started_at, ts)); + stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, ts)); } -/// Non-negative milliseconds between two instants, saturating at zero so a -/// clock skew or an out-of-order replay cannot produce a negative span. -fn elapsed_ms(from: DateTime, to: DateTime) -> u64 { - u64::try_from(to.signed_duration_since(from).num_milliseconds().max(0)).unwrap_or(0) -} - -/// Close every bracket opened by the session that just ended. +/// Close every active bracket opened by the session that just ended. /// /// `agent.session.ended` is the only ordering-safe backstop for terminal /// cancel and wall-clock timeout, which tear the session down through @@ -1147,7 +1149,7 @@ fn elapsed_ms(from: DateTime, to: DateTime) -> u64 { /// opened. Implemented as a normal stage lookup it would find no target and /// silently no-op, leaving the bracket open forever on exactly the path it /// exists to cover. -fn close_inference_brackets_for_session( +fn close_active_brackets_for_session( state: &mut RunProjection, stored: &RunEvent, ts: DateTime, @@ -1166,6 +1168,7 @@ fn close_inference_brackets_for_session( if opened_here { close_bracket_on_stage(stage, ts); } + stage.close_tool_batch_for_session(session_id, ts); } } @@ -1475,14 +1478,15 @@ fn finalize_unfinished_stages_after_run_failed( StageState::Failed }; - for (_, stage) in state.iter_stages_mut() { + for (_, stage) in state.iter_stages_unordered_mut() { if stage.state.is_terminal() { continue; } - // Close any bracket still open so its span is not dropped on the + // Close any brackets still open so their spans are not dropped on the // floor when the live estimate is frozen into `timing` below. close_bracket_on_stage(stage, timestamp); + stage.close_open_tool_batch(timestamp); // Freeze the live estimate before flipping to a terminal state: // `live_timing` reads `effective_state` and would return wall-only @@ -1490,7 +1494,9 @@ fn finalize_unfinished_stages_after_run_failed( let frozen = stage.live_timing(timestamp); stage.state = terminal_state; if stage.timing.is_none() && stage.started_at.is_some() { - stage.timing = Some(frozen); + stage.set_authoritative_timing(frozen); + } else { + stage.clear_live_timing(); } } } @@ -1641,12 +1647,16 @@ mod tests { StageId::new("plan", 1) } - fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope { + fn session_event(seq: u32, ts: &str, session_id: &str, body: EventBody) -> EventEnvelope { let mut event = test_stage_event_at(seq, ts, body, stage_id()); - event.event.session_id = Some("session-1".to_string()); + event.event.session_id = Some(session_id.to_string()); event } + fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope { + session_event(seq, ts, "session-1", body) + } + /// An event from a sub-agent session nested under the root session. fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope { let mut event = agent_event(seq, ts, body); @@ -1855,6 +1865,81 @@ mod tests { assert!(stage(&state).tool_batch.is_some()); } + #[test] + fn a_foreign_session_completion_does_not_mutate_the_open_batch() { + let mut state = started_state(); + state + .apply_event(&agent_event( + 2, + "2026-04-07T12:00:00Z", + tool_started("call-a"), + )) + .unwrap(); + state + .apply_event(&session_event( + 3, + "2026-04-07T12:00:05Z", + "session-2", + tool_completed("call-a"), + )) + .unwrap(); + + let batch = stage(&state).tool_batch.as_ref().unwrap(); + assert_eq!(batch.session_id, "session-1"); + assert!(batch.open_call_ids.contains("call-a")); + assert_eq!(stage(&state).live_tool_ms, 0); + } + + #[test] + fn a_replacement_session_starts_a_separate_tool_batch() { + let mut state = started_state(); + state + .apply_event(&agent_event( + 2, + "2026-04-07T12:00:00Z", + tool_started("old-call"), + )) + .unwrap(); + state + .apply_event(&session_event( + 3, + "2026-04-07T12:00:05Z", + "session-2", + tool_started("new-call"), + )) + .unwrap(); + + assert_eq!(stage(&state).live_tool_ms, 5_000); + let batch = stage(&state).tool_batch.as_ref().unwrap(); + assert_eq!(batch.session_id, "session-2"); + assert_eq!( + batch.open_call_ids, + ["new-call".to_string()].into_iter().collect() + ); + + // A delayed completion from the old session cannot close the new + // session's batch even when call ids happen to collide. + state + .apply_event(&agent_event( + 4, + "2026-04-07T12:00:07Z", + tool_completed("new-call"), + )) + .unwrap(); + assert!(stage(&state).tool_batch.is_some()); + + state + .apply_event(&session_event( + 5, + "2026-04-07T12:00:09Z", + "session-2", + tool_completed("new-call"), + )) + .unwrap(); + assert_eq!(stage(&state).live_tool_ms, 9_000); + assert!(stage(&state).tool_batch.is_none()); + } + #[test] fn subagent_tool_calls_do_not_double_count_against_the_root_batch() { let mut state = started_state(); @@ -1892,14 +1977,21 @@ mod tests { } #[test] - fn session_end_accumulates_rather_than_discarding_the_bracket() { + fn session_end_accumulates_every_open_active_bracket() { let mut state = started_state(); state .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started())) .unwrap(); + state + .apply_event(&agent_event( + 3, + "2026-04-07T12:00:07Z", + tool_started("call-a"), + )) + .unwrap(); let mut ended = test_stage_event_at( - 3, + 4, "2026-04-07T12:00:20Z", EventBody::AgentSessionEnded(AgentSessionEndedProps {}), stage_id(), @@ -1908,7 +2000,9 @@ mod tests { state.apply_event(&ended).unwrap(); assert_eq!(stage(&state).live_inference_ms, 15_000); + assert_eq!(stage(&state).live_tool_ms, 13_000); assert!(stage(&state).inference.is_none()); + assert!(stage(&state).tool_batch.is_none()); } #[test] @@ -1986,6 +2080,48 @@ mod tests { stage.timing.unwrap(), "a terminal stage reports its finalized breakdown, not a live estimate" ); + assert_eq!(stage.live_inference_ms, 0); + assert_eq!(stage.live_tool_ms, 0); + assert!(stage.inference.is_none()); + assert!(stage.tool_batch.is_none()); + } + + #[test] + fn run_failure_freezes_open_work_and_clears_live_bookkeeping() { + let mut state = started_state(); + state.status = RunStatus::Running; + state + .apply_event(&agent_event(2, "2026-04-07T12:00:01Z", llm_started())) + .unwrap(); + state + .apply_event(&agent_event(3, "2026-04-07T12:00:04Z", agent_message())) + .unwrap(); + state + .apply_event(&agent_event( + 4, + "2026-04-07T12:00:05Z", + tool_started("call-a"), + )) + .unwrap(); + + let mut failed = test_event( + 5, + EventBody::RunFailed(run_failed_props(FailureReason::WorkflowError)), + None, + ); + failed.event.ts = test_dt("2026-04-07T12:00:10Z"); + state.apply_event(&failed).unwrap(); + + let stage = stage(&state); + assert_eq!( + stage.timing, + Some(fabro_types::StageTiming::new(10_000, 3_000, 5_000)) + ); + assert_eq!(stage.state, StageState::Failed); + assert_eq!(stage.live_inference_ms, 0); + assert_eq!(stage.live_tool_ms, 0); + assert!(stage.inference.is_none()); + assert!(stage.tool_batch.is_none()); } } diff --git a/lib/foundation/fabro-api/build.rs b/lib/foundation/fabro-api/build.rs index fc943e99d..740585545 100644 --- a/lib/foundation/fabro-api/build.rs +++ b/lib/foundation/fabro-api/build.rs @@ -361,6 +361,11 @@ fn main() { "fabro_types::StageInferenceProjection", &[], ), + ( + "StageToolBatchProjection", + "fabro_types::StageToolBatchProjection", + &[], + ), ("LlmOutputKind", "fabro_types::LlmOutputKind", &[]), ("PermissionLevel", "fabro_types::PermissionLevel", &[]), ( diff --git a/lib/foundation/fabro-api/src/lib.rs b/lib/foundation/fabro-api/src/lib.rs index 4312acb91..b9a6a7185 100644 --- a/lib/foundation/fabro-api/src/lib.rs +++ b/lib/foundation/fabro-api/src/lib.rs @@ -69,10 +69,10 @@ pub mod types { StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason, StageContextWindowWarning, StageHandler, StageId, StageInferenceProjection, - StageModelUsage, StageOutcome, StageProjection, StageState, SubAgentProjection, - SubAgentStatus, SystemActorKind, SystemIntegrationStatus, SystemIntegrationsResponse, - TodoListProjection, TurnId, UpdateVariableRequest, UserPrincipal, Variable, - VariableListResponse, WorkflowSettings, + StageModelUsage, StageOutcome, StageProjection, StageState, StageToolBatchProjection, + SubAgentProjection, SubAgentStatus, SystemActorKind, SystemIntegrationStatus, + SystemIntegrationsResponse, TodoListProjection, TurnId, UpdateVariableRequest, + UserPrincipal, Variable, VariableListResponse, WorkflowSettings, }; pub use crate::generated::types::*; diff --git a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs index 5faed1bd3..22a544e69 100644 --- a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs +++ b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs @@ -18,6 +18,7 @@ use fabro_api::types::{ StageContextWindowUnavailableReason as ApiStageContextWindowUnavailableReason, StageContextWindowWarning as ApiStageContextWindowWarning, StageInferenceProjection as ApiStageInferenceProjection, StageProjection as ApiStageProjection, + StageToolBatchProjection as ApiStageToolBatchProjection, SubAgentProjection as ApiSubAgentProjection, SubAgentStatus as ApiSubAgentStatus, TodoListProjection as ApiTodoListProjection, }; @@ -29,8 +30,8 @@ use fabro_types::{ ParallelBranchResult, PermissionLevel, SkillsProjection, StageContextWindow, StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason, - StageContextWindowWarning, StageInferenceProjection, StageProjection, SubAgentProjection, - SubAgentStatus, TodoListKind, TodoListProjection, + StageContextWindowWarning, StageInferenceProjection, StageProjection, StageToolBatchProjection, + SubAgentProjection, SubAgentStatus, TodoListKind, TodoListProjection, }; use serde_json::json; @@ -68,9 +69,32 @@ fn stage_projection_reuses_nested_agent_state_types() { ); assert_same_type::(); assert_same_type::(); + assert_same_type::(); assert_same_type::(); } +#[test] +fn stage_tool_batch_projection_matches_openapi_json_shape() { + let batch = StageToolBatchProjection { + session_id: "ses_root".to_string(), + started_at: "2026-04-29T12:34:00Z".parse().unwrap(), + open_call_ids: ["call_1".to_string(), "call_2".to_string()] + .into_iter() + .collect(), + }; + let value = serde_json::to_value(&batch).unwrap(); + assert_eq!( + value, + json!({ + "session_id": "ses_root", + "started_at": "2026-04-29T12:34:00Z", + "open_call_ids": ["call_1", "call_2"] + }) + ); + let api_batch: ApiStageToolBatchProjection = serde_json::from_value(value).unwrap(); + assert_eq!(api_batch, batch); +} + #[test] fn stage_inference_projection_matches_openapi_json_shape() { let inference = StageInferenceProjection { @@ -148,6 +172,9 @@ fn stage_projection_without_inference_round_trips() { let stage: StageProjection = serde_json::from_value(value.clone()).unwrap(); assert!(stage.inference.is_none()); + assert!(stage.tool_batch.is_none()); + assert_eq!(stage.live_inference_ms, 0); + assert_eq!(stage.live_tool_ms, 0); assert_eq!(serde_json::to_value(stage).unwrap(), value); } diff --git a/lib/foundation/fabro-types/src/lib.rs b/lib/foundation/fabro-types/src/lib.rs index 68f5644a2..b7f3b0474 100644 --- a/lib/foundation/fabro-types/src/lib.rs +++ b/lib/foundation/fabro-types/src/lib.rs @@ -125,7 +125,7 @@ pub use run_projection::{ StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason, StageContextWindowWarning, StageInferenceProjection, StageModelUsage, StageProjection, - SubAgentProjection, SubAgentStatus, first_event_seq, + StageToolBatchProjection, SubAgentProjection, SubAgentStatus, first_event_seq, }; pub use run_sandbox::{ RunSandbox, RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan, diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs index 0576d32f6..e2b9ad3f6 100644 --- a/lib/foundation/fabro-types/src/run_projection.rs +++ b/lib/foundation/fabro-types/src/run_projection.rs @@ -12,7 +12,7 @@ use crate::{ AgentToolSummary, BilledTokenCounts, Checkpoint, Conclusion, InterviewQuestionRecord, InvalidTransition, LlmOutputKind, ModelRef, PermissionLevel, PullRequestLink, RunApproval, RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, RunTiming, StageCompletion, - StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection, + StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection, timing, }; #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] @@ -433,6 +433,10 @@ fn is_zero_ms(value: &u64) -> bool { /// duplicated log must not let a repeated completion drain the batch early. #[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)] pub struct StageToolBatchProjection { + /// Root agent session that dispatched the batch. Later transitions are + /// gated on it so delayed events from a replaced session cannot mutate + /// the current session's batch. + pub session_id: String, /// When the batch opened — the first `agent.tool.started` observed while /// no other calls were outstanding. pub started_at: DateTime, @@ -603,10 +607,9 @@ impl StageProjection { state, StageState::Running | StageState::Retrying | StageState::Pending ) { - return self.started_at.map(|started| { - u64::try_from(now.signed_duration_since(started).num_milliseconds().max(0)) - .unwrap_or(0) - }); + return self + .started_at + .map(|started| timing::elapsed_ms(started, now)); } self.timing.map(|timing| timing.wall_time_ms) } @@ -645,9 +648,6 @@ impl StageProjection { } let wall_time_ms = self.live_wall_time_ms(now).unwrap_or(0); - let elapsed = |since: DateTime| { - u64::try_from(now.signed_duration_since(since).num_milliseconds().max(0)).unwrap_or(0) - }; // `handler` is absent on projections built from events written before // stage execution identity. Treat those as agent stages, matching @@ -660,11 +660,11 @@ impl StageProjection { let open_inference = self .inference .as_ref() - .map_or(0, |inference| elapsed(inference.started_at)); + .map_or(0, |inference| timing::elapsed_ms(inference.started_at, now)); let open_tool = self .tool_batch .as_ref() - .map_or(0, |batch| elapsed(batch.started_at)); + .map_or(0, |batch| timing::elapsed_ms(batch.started_at, now)); ( self.live_inference_ms.saturating_add(open_inference), self.live_tool_ms.saturating_add(open_tool), @@ -693,9 +693,26 @@ impl StageProjection { } /// Record a dispatched tool call, opening a batch if none is outstanding. - pub fn open_tool_call(&mut self, tool_call_id: String, started_at: DateTime) { + /// + /// If a replacement root session starts work before the old session's end + /// event arrives, freeze the old batch at this boundary before opening + /// the new one. This keeps the sessions separate without dropping time. + pub fn open_tool_call( + &mut self, + session_id: String, + tool_call_id: String, + started_at: DateTime, + ) { + let replaces_open_batch = self + .tool_batch + .as_ref() + .is_some_and(|batch| batch.session_id != session_id); + if replaces_open_batch { + self.close_open_tool_batch(started_at); + } self.tool_batch .get_or_insert_with(|| StageToolBatchProjection { + session_id, started_at, open_call_ids: BTreeSet::new(), }) @@ -705,21 +722,52 @@ impl StageProjection { /// Retire a tool call. Folds the batch into the live accumulator once the /// last outstanding call reports, so concurrent calls count once. - pub fn close_tool_call(&mut self, tool_call_id: &str, now: DateTime) { + pub fn close_tool_call(&mut self, session_id: &str, tool_call_id: &str, now: DateTime) { let Some(batch) = self.tool_batch.as_mut() else { return; }; - batch.open_call_ids.remove(tool_call_id); - if !batch.open_call_ids.is_empty() { + if batch.session_id != session_id + || !batch.open_call_ids.remove(tool_call_id) + || !batch.open_call_ids.is_empty() + { return; } - let elapsed = u64::try_from( - now.signed_duration_since(batch.started_at) - .num_milliseconds() - .max(0), - ) - .unwrap_or(0); - self.live_tool_ms = self.live_tool_ms.saturating_add(elapsed); + self.close_open_tool_batch(now); + } + + /// Close a tool batch only when it belongs to `session_id`. + pub fn close_tool_batch_for_session(&mut self, session_id: &str, now: DateTime) { + let opened_here = self + .tool_batch + .as_ref() + .is_some_and(|batch| batch.session_id == session_id); + if opened_here { + self.close_open_tool_batch(now); + } + } + + /// Fold any open tool batch into the live accumulator. + pub fn close_open_tool_batch(&mut self, now: DateTime) { + let Some(batch) = self.tool_batch.take() else { + return; + }; + self.live_tool_ms = self + .live_tool_ms + .saturating_add(timing::elapsed_ms(batch.started_at, now)); + } + + /// Install a worker-provided terminal timing and discard transient live + /// bookkeeping that is no longer authoritative. + pub fn set_authoritative_timing(&mut self, timing: StageTiming) { + self.timing = Some(timing); + self.clear_live_timing(); + } + + /// Discard transient timing accumulators and open brackets. + pub fn clear_live_timing(&mut self) { + self.live_inference_ms = 0; + self.live_tool_ms = 0; + self.inference = None; self.tool_batch = None; } @@ -1138,20 +1186,15 @@ mod iter_stages_tests { #[cfg(test)] mod live_timing_tests { use std::collections::HashMap; - use std::num::NonZeroU32; use chrono::{DateTime, TimeZone, Utc}; use super::{RunProjection, StageToolBatchProjection}; use crate::{ Graph, ModelRef, RunId, RunSpec, StageHandler, StageInferenceProjection, StageProjection, - StageState, StageTiming, StartRecord, WorkflowSettings, test_support, + StageState, StageTiming, StartRecord, WorkflowSettings, first_event_seq, test_support, }; - fn seq(n: u32) -> NonZeroU32 { - NonZeroU32::new(n).unwrap() - } - fn at(seconds: i64) -> DateTime { Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap() } @@ -1180,7 +1223,7 @@ mod live_timing_tests { /// In-flight stage that started at `at(0)`. fn running(handler: StageHandler) -> StageProjection { - let mut stage = StageProjection::new(seq(1)); + let mut stage = StageProjection::new(first_event_seq(1)); stage.handler = Some(handler); stage.started_at = Some(at(0)); stage.state = StageState::Running; @@ -1225,6 +1268,7 @@ mod live_timing_tests { // Inference open for 20s, tools open for 10s, at t=120s. stage.inference = Some(open_bracket(at(100))); stage.tool_batch = Some(StageToolBatchProjection { + session_id: "session-1".to_string(), started_at: at(110), open_call_ids: ["call-1".to_string()].into_iter().collect(), }); @@ -1330,17 +1374,17 @@ mod live_timing_tests { base_sha: None, }); - let baseline = projection.stage_entry("baseline", 1, seq(1)); + let baseline = projection.stage_entry("baseline", 1, first_event_seq(1)); baseline.handler = Some(StageHandler::Command); baseline.state = StageState::Succeeded; baseline.timing = Some(StageTiming::new(42_666, 0, 42_663)); - let assess = projection.stage_entry("assess", 1, seq(2)); + let assess = projection.stage_entry("assess", 1, first_event_seq(2)); assess.handler = Some(StageHandler::Agent); assess.state = StageState::Succeeded; assess.timing = Some(StageTiming::new(86_025, 78_230, 7_588)); - let plan = projection.stage_entry("plan", 1, seq(3)); + let plan = projection.stage_entry("plan", 1, first_event_seq(3)); plan.handler = Some(StageHandler::Agent); plan.started_at = Some(at(146)); plan.state = StageState::Running; @@ -1367,7 +1411,8 @@ mod live_timing_tests { }); for (index, node) in ["branch-a", "branch-b", "branch-c"].iter().enumerate() { - let stage = projection.stage_entry(node, 1, seq(u32::try_from(index).unwrap() + 1)); + let stage = + projection.stage_entry(node, 1, first_event_seq(u32::try_from(index).unwrap() + 1)); stage.handler = Some(StageHandler::Agent); stage.state = StageState::Succeeded; stage.timing = Some(StageTiming::new(60_000, 60_000, 0)); diff --git a/lib/foundation/fabro-types/src/timing.rs b/lib/foundation/fabro-types/src/timing.rs index 4fe45a36f..812a72ef2 100644 --- a/lib/foundation/fabro-types/src/timing.rs +++ b/lib/foundation/fabro-types/src/timing.rs @@ -18,6 +18,16 @@ use chrono::{DateTime, Utc}; use serde::{Deserialize, Serialize}; +/// Non-negative milliseconds between two instants. +/// +/// Clock skew or an out-of-order replay can put `end` before `start`; those +/// spans contribute zero rather than wrapping into a large unsigned value. +#[must_use] +pub fn elapsed_ms(start: DateTime, end: DateTime) -> u64 { + u64::try_from(end.signed_duration_since(start).num_milliseconds().max(0)) + .expect("non-negative chrono millisecond durations fit in u64") +} + /// Timing breakdown for one stage visit. #[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct StageTiming { @@ -74,16 +84,18 @@ impl StageTiming { /// legitimately sum past run wall time. #[must_use] pub fn clamped_to_wall(&self) -> Self { - if self.active_time_ms <= self.wall_time_ms { + let active_time_ms = u128::from(self.inference_time_ms) + u128::from(self.tool_time_ms); + if active_time_ms <= u128::from(self.wall_time_ms) { return *self; } // Preserve the split rather than truncating one side, so a clamped // stage still shows where its time went. Widen for the multiply: the - // quotient is bounded by `wall_time_ms` because `active_time_ms` - // exceeds it here, so it always fits back into u64. - let scaled = u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) - / u128::from(self.active_time_ms); - let inference_time_ms = u64::try_from(scaled).unwrap_or(self.wall_time_ms); + // quotient is bounded by `wall_time_ms` because the exact, widened + // active total exceeds it here, so it always fits back into u64. + let scaled = + u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) / active_time_ms; + let inference_time_ms = + u64::try_from(scaled).expect("scaled inference time is bounded by wall time"); let tool_time_ms = self.wall_time_ms.saturating_sub(inference_time_ms); Self::new(self.wall_time_ms, inference_time_ms, tool_time_ms) } @@ -166,8 +178,7 @@ impl RunTiming { /// Milliseconds elapsed from `start` to `now`, clamped at zero. #[must_use] pub fn wall_time_ms_since(start: DateTime, now: DateTime) -> u64 { - u64::try_from(now.signed_duration_since(start).num_milliseconds().max(0)) - .expect("non-negative milliseconds fit in u64") + elapsed_ms(start, now) } } @@ -184,7 +195,9 @@ impl From for RunTiming { #[cfg(test)] mod tests { - use super::{RunTiming, StageTiming}; + use chrono::{TimeZone, Utc}; + + use super::{RunTiming, StageTiming, elapsed_ms}; #[test] fn stage_timing_new_derives_active_as_sum_of_inference_and_tool() { @@ -215,6 +228,23 @@ mod tests { assert_eq!(sum.active_time_ms, 175); } + #[test] + fn stage_timing_clamp_uses_the_unsaturated_active_total() { + let timing = StageTiming::new(u64::MAX, u64::MAX, u64::MAX).clamped_to_wall(); + + assert_eq!(timing.inference_time_ms, u64::MAX / 2); + assert_eq!(timing.tool_time_ms, u64::MAX.saturating_sub(u64::MAX / 2)); + assert_eq!(timing.active_time_ms, u64::MAX); + } + + #[test] + fn elapsed_ms_clamps_out_of_order_instants_to_zero() { + let later = Utc.timestamp_opt(100, 0).unwrap(); + let earlier = Utc.timestamp_opt(99, 0).unwrap(); + + assert_eq!(elapsed_ms(later, earlier), 0); + } + #[test] fn run_timing_wall_only_zeroes_breakdown_and_active() { let timing = RunTiming::wall_only(1500); diff --git a/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts index a6bd7718c..c382b5dd1 100644 --- a/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts +++ b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts @@ -18,6 +18,10 @@ * One open tool batch: tool calls dispatched together that have not all reported completion. `open_call_ids` is a set rather than a count so a duplicated completion in a replayed log cannot drain the batch early. */ export interface StageToolBatchProjection { + /** + * Root agent session that dispatched the batch. Transitions are gated on it so delayed events from a replaced session cannot mutate the current batch. + */ + 'session_id': string; /** * When the batch opened — the first dispatched call observed while no other calls were outstanding. */ @@ -25,5 +29,5 @@ export interface StageToolBatchProjection { /** * Calls dispatched but not yet completed, by tool call id. */ - 'open_call_ids': Array; + 'open_call_ids': Set; } From f293e3de18af896440d2d810699bb840728ac2c4 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 23:44:20 -0400 Subject: [PATCH 05/76] refactor(agent): fold parent notifications into subagent state `ParentNotificationHub` kept a second `Mutex` and `watch` channel holding a copy of each child's terminal result -- data `SubAgent.status` already owns as `SubAgentStatus::Finished`, and which is never evicted, since nothing removes entries from `SupervisorState.agents`. Two of the three bugs fixed in the previous commit were ordering bugs in the coupling between those two structures: suppress-vs-commit in `begin_shutdown`, and register-vs-publish in `spawn_inner`. Both were fixed by ordering the steps correctly. Keeping the registration beside the status it is delivered with makes that whole class unrepresentable instead: - Registration is now a field on the `SubAgent` literal `spawn_inner` already builds, under the lock that publishes it. There is no window between publishing an agent and registering its notification. - Suppression on shutdown happens inside the critical section that decides the shutdown, after the status transition commits, so a rejected shutdown cannot discard a result the parent is owed. - `next_parent_notification_batch` scans agents for a live registration whose status is `Finished`, and ignores `Closing`/`Closed` outright -- so a shutdown racing delivery can no longer park the parent on a result that will never arrive, even if suppression were missed. `spawn_result_monitor` no longer takes the hub; it bumps a single `watch` counter after committing the status it already commits. Batch order was the queue's insertion order, so `SubAgent` carries a `spawn_seq` to keep delivery oldest-first. Tests move from exercising the hub directly to the supervisor API, and cover spawn-order batching and the shutdown-races-delivery case that the old shape could not express. Co-Authored-By: Claude Opus 5 (1M context) --- lib/components/fabro-agent/src/subagent.rs | 446 ++++++++++++--------- 1 file changed, 259 insertions(+), 187 deletions(-) diff --git a/lib/components/fabro-agent/src/subagent.rs b/lib/components/fabro-agent/src/subagent.rs index f4d6c1b84..476094a94 100644 --- a/lib/components/fabro-agent/src/subagent.rs +++ b/lib/components/fabro-agent/src/subagent.rs @@ -89,16 +89,27 @@ pub enum SubAgentStatus { const SUBAGENT_SHUTDOWN_GRACE: Duration = Duration::from_secs(5); struct SubAgent { - status: watch::Sender, - cleanup_done: watch::Sender, - cleanup_started: bool, - monitor_task: Option>, - event_forwarder: Option>, - cleanup_task: Option>, - child_abort_handle: AbortHandle, - followup_queue: Arc>>, - cancel_token: CancellationToken, - depth: usize, + status: watch::Sender, + cleanup_done: watch::Sender, + cleanup_started: bool, + monitor_task: Option>, + event_forwarder: Option>, + cleanup_task: Option>, + child_abort_handle: AbortHandle, + followup_queue: Arc>>, + cancel_token: CancellationToken, + depth: usize, + /// Task description, set when the parent should receive this child's + /// terminal result automatically. Cleared once the result is delivered, + /// the parent retrieves it explicitly, or the agent is shut down. + /// + /// Keeping this beside the status it is delivered with means a + /// notification cannot be registered before -- or suppressed after -- the + /// state it describes: there is only one lock and one ordering. + parent_notification: Option, + /// Spawn order, so a batch is delivered oldest-first rather than in + /// whatever order the map happens to iterate. + spawn_seq: u64, } impl Drop for SubAgent { @@ -119,114 +130,8 @@ impl Drop for SubAgent { #[derive(Default)] struct SupervisorState { - agents: HashMap, -} - -#[derive(Default)] -struct ParentNotificationState { - pending: HashMap, - ready: VecDeque, -} - -struct ParentNotificationHub { - state: Mutex, - changed: watch::Sender, -} - -impl ParentNotificationHub { - fn new() -> Self { - let (changed, _) = watch::channel(0); - Self { - state: Mutex::new(ParentNotificationState::default()), - changed, - } - } - - fn register(&self, agent_id: String, description: String) { - self.state - .lock() - .expect("parent notification lock poisoned") - .pending - .insert(agent_id, description); - self.signal(); - } - - fn complete(&self, agent_id: &str, result: Result) { - { - let mut state = self - .state - .lock() - .expect("parent notification lock poisoned"); - let Some(description) = state.pending.remove(agent_id) else { - return; - }; - state.ready.push_back(SubAgentParentNotification { - agent_id: agent_id.to_string(), - description, - result, - }); - } - self.signal(); - } - - fn suppress(&self, agent_id: &str) { - let changed = { - let mut state = self - .state - .lock() - .expect("parent notification lock poisoned"); - let removed_pending = state.pending.remove(agent_id).is_some(); - let ready_len = state.ready.len(); - state - .ready - .retain(|notification| notification.agent_id != agent_id); - removed_pending || state.ready.len() != ready_len - }; - if changed { - self.signal(); - } - } - - async fn next_batch( - &self, - cancel: &CancellationToken, - ) -> Result>, Error> { - let mut changed = self.changed.subscribe(); - loop { - { - let mut state = self - .state - .lock() - .expect("parent notification lock poisoned"); - if !state.ready.is_empty() { - return Ok(Some(state.ready.drain(..).collect())); - } - if state.pending.is_empty() { - return Ok(None); - } - } - - tokio::select! { - biased; - () = cancel.cancelled() => { - return Err(Error::Interrupted(InterruptReason::Cancelled)); - } - observed = changed.changed() => { - observed.map_err(|_| { - Error::InvalidState( - "Background-agent notification observer closed unexpectedly".to_string(), - ) - })?; - } - } - } - } - - fn signal(&self) { - self.changed.send_modify(|generation| { - *generation = generation.wrapping_add(1); - }); - } + agents: HashMap, + next_spawn_seq: u64, } struct ShutdownWork { @@ -268,11 +173,20 @@ impl Drop for CleanupDoneGuard { } } +/// Wake anything parked in +/// [`SubAgentSupervisor::next_parent_notification_batch`] so it can re-evaluate +/// which children are deliverable. +fn signal_notifications(changed: &watch::Sender) { + changed.send_modify(|generation| { + *generation = generation.wrapping_add(1); + }); +} + fn spawn_result_monitor( child_task: JoinHandle>, status: watch::Sender, event_callback: Arc>>, - parent_notifications: Arc, + notifications_changed: Arc>, agent_id: String, depth: usize, ) -> JoinHandle<()> { @@ -294,6 +208,8 @@ fn spawn_result_monitor( if !committed { return; } + // The status this agent will be delivered with is now committed. + signal_notifications(¬ifications_changed); let event = match &task_result { Ok(result) => AgentEvent::SubAgentCompleted { @@ -315,7 +231,6 @@ fn spawn_result_monitor( if let Some(callback) = callback { callback(SubAgentCallbackEvent::Lifecycle(event)); } - parent_notifications.complete(&agent_id, task_result); }) } @@ -326,10 +241,10 @@ fn spawn_result_monitor( /// happen after the guard has been released. #[derive(Clone)] pub struct SubAgentSupervisor { - state: Arc>, - max_depth: usize, - event_callback: Arc>>, - parent_notifications: Arc, + state: Arc>, + max_depth: usize, + event_callback: Arc>>, + notifications_changed: Arc>, } impl SubAgentSupervisor { @@ -339,7 +254,7 @@ impl SubAgentSupervisor { state: Arc::new(Mutex::new(SupervisorState::default())), max_depth, event_callback: Arc::new(RwLock::new(None)), - parent_notifications: Arc::new(ParentNotificationHub::new()), + notifications_changed: Arc::new(watch::channel(0).0), } } @@ -474,23 +389,15 @@ impl SubAgentSupervisor { child_task, status.clone(), Arc::clone(&self.event_callback), - Arc::clone(&self.parent_notifications), + Arc::clone(&self.notifications_changed), agent_id.clone(), child_depth, ); - // Register before the agent becomes discoverable in `state.agents`. - // Once it is, a concurrent `shutdown_all` can suppress and close it; - // registering afterwards would leave a pending entry that the monitor - // never completes (it early-returns for a non-Running agent), and - // `next_batch` would then never report the queue as drained. Nothing - // can complete this registration before `start_tx.send(())` below. - if let Some(description) = parent_notification_description { - self.parent_notifications - .register(agent_id.clone(), description); - } { let mut state = self.state.lock().expect("subagent state lock poisoned"); + let spawn_seq = state.next_spawn_seq; + state.next_spawn_seq = state.next_spawn_seq.saturating_add(1); state.agents.insert(agent_id.clone(), SubAgent { status, cleanup_done, @@ -502,8 +409,11 @@ impl SubAgentSupervisor { followup_queue, cancel_token, depth: child_depth, + parent_notification: parent_notification_description, + spawn_seq, }); } + signal_notifications(&self.notifications_changed); self.emit_event(AgentEvent::SubAgentSpawned { agent_id: agent_id.clone(), @@ -587,11 +497,20 @@ impl SubAgentSupervisor { } } - /// Stop automatic delivery for an agent whose result the parent explicitly - /// retrieved. Removes a result that may already have raced into the ready - /// queue. + /// Stop automatic delivery for an agent whose result the parent retrieved + /// explicitly. pub(crate) fn suppress_parent_notification(&self, agent_id: &str) { - self.parent_notifications.suppress(agent_id); + let cleared = { + let mut state = self.state.lock().expect("subagent state lock poisoned"); + state + .agents + .get_mut(agent_id) + .and_then(|agent| agent.parent_notification.take()) + .is_some() + }; + if cleared { + signal_notifications(&self.notifications_changed); + } } /// Wait until all currently-ready background results can be delivered in @@ -616,7 +535,68 @@ impl SubAgentSupervisor { &self, cancel: &CancellationToken, ) -> Result>, Error> { - self.parent_notifications.next_batch(cancel).await + let mut changed = self.notifications_changed.subscribe(); + loop { + { + let mut state = self.state.lock().expect("subagent state lock poisoned"); + let mut ready = Vec::new(); + let mut awaiting_result = false; + for (agent_id, agent) in &state.agents { + let Some(description) = agent.parent_notification.as_ref() else { + continue; + }; + let finished = match &*agent.status.borrow() { + SubAgentStatus::Finished(result) => Some(result.clone()), + SubAgentStatus::Running => { + awaiting_result = true; + None + } + // Being torn down, so no result is coming. Ignoring + // these is what keeps a shutdown that races delivery + // from parking the parent forever. + SubAgentStatus::Closing | SubAgentStatus::Closed => None, + }; + if let Some(result) = finished { + ready.push((agent.spawn_seq, SubAgentParentNotification { + agent_id: agent_id.clone(), + description: description.clone(), + result, + })); + } + } + + if !ready.is_empty() { + ready.sort_by_key(|(spawn_seq, _)| *spawn_seq); + let batch: Vec<_> = ready + .into_iter() + .map(|(_, notification)| notification) + .collect(); + for notification in &batch { + if let Some(agent) = state.agents.get_mut(¬ification.agent_id) { + agent.parent_notification = None; + } + } + return Ok(Some(batch)); + } + if !awaiting_result { + return Ok(None); + } + } + + tokio::select! { + biased; + () = cancel.cancelled() => { + return Err(Error::Interrupted(InterruptReason::Cancelled)); + } + observed = changed.changed() => { + observed.map_err(|_| { + Error::InvalidState( + "Background-agent notification observer closed unexpectedly".to_string(), + ) + })?; + } + } + } } #[cfg(test)] @@ -666,6 +646,11 @@ impl SubAgentSupervisor { } }; + // Shutdown is committed, so this child's result will never reach the + // parent. The early returns above leave the notification intact, so a + // rejected shutdown cannot discard a result the parent is owed. + agent.parent_notification = None; + if agent.cleanup_started { return Ok(ShutdownDisposition::Follow(agent.cleanup_done.subscribe())); } @@ -772,10 +757,7 @@ impl SubAgentSupervisor { async fn ensure_closed(&self, agent_id: &str) -> Result<(), Error> { let disposition = self.begin_shutdown(agent_id, false)?; - // Only once shutdown is committed. Suppressing before `begin_shutdown` - // would also discard the result of an agent that had already finished, - // which rejects the shutdown but had a delivery pending. - self.parent_notifications.suppress(agent_id); + signal_notifications(&self.notifications_changed); let cleanup_done = match disposition { ShutdownDisposition::Lead(work) => self.spawn_shutdown(work), ShutdownDisposition::Follow(cleanup_done) => cleanup_done, @@ -788,7 +770,7 @@ impl SubAgentSupervisor { /// Strict user-facing close: only a currently running child may be closed. pub async fn close_agent(&self, agent_id: &str) -> Result<(), Error> { let disposition = self.begin_shutdown(agent_id, true)?; - self.parent_notifications.suppress(agent_id); + signal_notifications(&self.notifications_changed); let cleanup_done = match disposition { ShutdownDisposition::Lead(work) => self.spawn_shutdown(work), ShutdownDisposition::Follow(_) | ShutdownDisposition::Done => { @@ -857,7 +839,7 @@ impl SubAgentSupervisor { child_task, status.clone(), Arc::clone(&self.event_callback), - Arc::clone(&self.parent_notifications), + Arc::clone(&self.notifications_changed), agent_id.clone(), depth, ); @@ -873,6 +855,8 @@ impl SubAgentSupervisor { event_forwarder, cleanup_task: None, child_abort_handle, + parent_notification: None, + spawn_seq: 0, followup_queue: Arc::new(Mutex::new(VecDeque::new())), cancel_token, depth, @@ -1072,67 +1056,155 @@ mod tests { assert!(manager.is_empty()); } - #[tokio::test] - async fn parent_notifications_are_exactly_once_and_xml_escaped() { - let hub = ParentNotificationHub::new(); - hub.register("agent<&".to_string(), "Review & tests".to_string()); - let result = Ok(SubAgentResult { - output: "done & \"verified\"".to_string(), - success: true, - turns_used: 2, - }); - hub.complete("agent<&", result.clone()); - hub.complete("agent<&", result); + #[test] + fn parent_notification_envelope_escapes_xml() { + let envelope = format_parent_notification_batch(&[SubAgentParentNotification { + agent_id: "agent<&".to_string(), + description: "Review & tests".to_string(), + result: Ok(SubAgentResult { + output: "done & \"verified\"".to_string(), + success: true, + turns_used: 2, + }), + }]); - let notifications = hub - .next_batch(&CancellationToken::new()) - .await - .unwrap() - .unwrap(); - assert_eq!(notifications.len(), 1); - let envelope = format_parent_notification_batch(¬ifications); assert!(envelope.contains("completed")); assert!(envelope.contains("agent<&")); assert!(envelope.contains("Review <core> & tests")); assert!( envelope.contains("done <safely> & "verified"") ); - assert!( - hub.next_batch(&CancellationToken::new()) - .await - .unwrap() - .is_none() - ); } #[tokio::test] - async fn suppress_removes_pending_and_ready_parent_notifications() { - let hub = ParentNotificationHub::new(); - hub.register("pending".to_string(), "Pending".to_string()); - hub.suppress("pending"); + async fn a_finished_agent_is_delivered_to_the_parent_exactly_once() { + let supervisor = SubAgentSupervisor::new(3); + let child = make_session(vec![text_response("child result")]).await; + let agent_id = supervisor + .spawn_with_parent_notification( + child, + "task".to_string(), + "Inspect the module".to_string(), + 0, + ) + .unwrap(); + supervisor + .wait_with_cancel(&agent_id, &CancellationToken::new()) + .await + .unwrap(); + + let batch = supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .expect("the finished child must be delivered"); + assert_eq!(batch.len(), 1); + assert_eq!(batch[0].agent_id, agent_id); + assert_eq!(batch[0].description, "Inspect the module"); + + // The status stays `Finished`, so re-delivery is prevented by clearing + // the registration rather than by consuming the result. assert!( - hub.next_batch(&CancellationToken::new()) + supervisor + .next_parent_notification_batch(&CancellationToken::new()) .await .unwrap() .is_none() ); - hub.register("ready".to_string(), "Ready".to_string()); - hub.complete( - "ready", - Ok(SubAgentResult { - output: "done".to_string(), - success: true, - turns_used: 1, - }), - ); - hub.suppress("ready"); + supervisor.shutdown_all().await; + } + + #[tokio::test] + async fn batches_are_delivered_in_spawn_order() { + let supervisor = SubAgentSupervisor::new(3); + let mut ids = Vec::new(); + for index in 0..3 { + let child = make_session(vec![text_response("done")]).await; + ids.push( + supervisor + .spawn_with_parent_notification( + child, + format!("task {index}"), + format!("Task {index}"), + 0, + ) + .unwrap(), + ); + } + for id in &ids { + supervisor + .wait_with_cancel(id, &CancellationToken::new()) + .await + .unwrap(); + } + + let batch = supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .expect("all three children must be delivered together"); + let delivered: Vec<_> = batch.iter().map(|n| n.agent_id.clone()).collect(); + assert_eq!(delivered, ids); + + supervisor.shutdown_all().await; + } + + #[tokio::test] + async fn suppressing_before_completion_stops_delivery() { + let supervisor = SubAgentSupervisor::new(3); + let child = make_session(vec![text_response("child result")]).await; + let agent_id = supervisor + .spawn_with_parent_notification( + child, + "task".to_string(), + "Inspect the module".to_string(), + 0, + ) + .unwrap(); + + supervisor.suppress_parent_notification(&agent_id); + supervisor + .wait_with_cancel(&agent_id, &CancellationToken::new()) + .await + .unwrap(); + assert!( - hub.next_batch(&CancellationToken::new()) + supervisor + .next_parent_notification_batch(&CancellationToken::new()) .await .unwrap() .is_none() ); + + supervisor.shutdown_all().await; + } + + #[tokio::test] + async fn closing_a_running_agent_stops_delivery_without_parking_the_parent() { + let supervisor = SubAgentSupervisor::new(3); + let child = make_session(vec![text_response("child result")]).await; + let agent_id = supervisor + .spawn_with_parent_notification( + child, + "task".to_string(), + "Inspect the module".to_string(), + 0, + ) + .unwrap(); + + supervisor.close_agent(&agent_id).await.unwrap(); + + // Must resolve rather than wait for a result that will never arrive. + assert!( + supervisor + .next_parent_notification_batch(&CancellationToken::new()) + .await + .unwrap() + .is_none() + ); + + supervisor.shutdown_all().await; } #[tokio::test] From 27549c2358edc12a20f25d732d80d3934b2955f2 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 26 Jul 2026 07:37:48 -0400 Subject: [PATCH 06/76] fix(agent): share the task runtime with every profile's children Task tools scope their list by `root_session_id` -- `Session` documents this as "a subagent session inherits its parent's `root_session_id` so todo tools that scope by root (Anthropic tasks) share one list across all subagents" -- so a root and its children address one logical list. `build()` runs once per session, though, and `AnthropicProfile` constructed its own `TodoRuntime` inside that call. Root and child therefore resolved the same `list_id` through different runtimes: both ID counters started at zero, so both emitted `todo.created` with id `1` for the same list, and `TodoListProjection::upsert` matches on id -- the child's task replaced the parent's in the persisted projection. `TaskGet` and `TaskList` read the local runtime, so neither session could see the other's tasks either. The previous commit's shared runtime fixed this for Claude 5 only, because `build()` passed dependencies positionally and adding a fourth argument would have meant touching all six call sites. It grew a second constructor for Claude 5 instead, leaving the other five on a signature that could not carry the runtime. Bundle them into `ProfileDeps` so every profile takes the same `(model, &deps)`. The duplicate constructor is gone, Anthropic shares the runtime by construction rather than by opting in, and a future dependency reaches all six profiles or none. The existing Claude 5 sharing test is generalized and now also runs for Anthropic; it fails against a per-profile runtime. Co-Authored-By: Claude Opus 5 (1M context) --- .../fabro-agent/src/profiles/anthropic.rs | 26 ++- .../fabro-agent/src/profiles/claude5.rs | 32 +--- .../fabro-agent/src/profiles/gemini.rs | 19 +-- .../fabro-agent/src/profiles/gpt56.rs | 13 +- .../fabro-agent/src/profiles/kimi.rs | 17 +- .../fabro-agent/src/profiles/mod.rs | 149 ++++++++++++------ .../fabro-agent/src/profiles/openai.rs | 17 +- 7 files changed, 148 insertions(+), 125 deletions(-) diff --git a/lib/components/fabro-agent/src/profiles/anthropic.rs b/lib/components/fabro-agent/src/profiles/anthropic.rs index 7cf84a80b..a34358d81 100644 --- a/lib/components/fabro-agent/src/profiles/anthropic.rs +++ b/lib/components/fabro-agent/src/profiles/anthropic.rs @@ -5,17 +5,14 @@ use fabro_model::{AgentProfileKind, Catalog, ProviderId}; use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; use crate::sandbox::Sandbox; use crate::skills::Skill; -use crate::todo_runtime::TodoRuntime; use crate::todo_tools::{ make_task_create_tool, make_task_get_tool, make_task_list_tool, make_task_update_tool, }; use crate::tool_registry::ToolRegistry; -use crate::tools::{ - WEB_SEARCH_TOOL_NAME, WebFetchSummarizer, make_edit_file_tool, register_core_tools, -}; +use crate::tools::{WEB_SEARCH_TOOL_NAME, make_edit_file_tool, register_core_tools}; pub struct AnthropicProfile { base: BaseProfile, @@ -26,21 +23,20 @@ const CORE_PROMPT: &str = include_str!("prompts/anthropic.md.j2"); impl AnthropicProfile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::Anthropic); - Self::with_native_tools(model, &options, None) + let deps = + ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::Anthropic)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - ) -> Self { + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { let mut registry = ToolRegistry::new(); - register_core_tools(&mut registry, options, summarizer); + register_core_tools(&mut registry, &deps.options, deps.summarizer.clone()); registry.register(make_edit_file_tool()); - // Anthropic task tools share one runtime per profile instance. - let todo_runtime = Arc::new(TodoRuntime::new()); + // Task tools scope their list by `root_session_id`, so a root session + // and its children address one logical list. They must therefore + // resolve it through the one runtime the builder shares between them. + let todo_runtime = Arc::clone(&deps.todo_runtime); registry.register(make_task_create_tool(todo_runtime.clone())); registry.register(make_task_update_tool(todo_runtime.clone())); registry.register(make_task_get_tool(todo_runtime.clone())); diff --git a/lib/components/fabro-agent/src/profiles/claude5.rs b/lib/components/fabro-agent/src/profiles/claude5.rs index d6d2faa6c..43a2db179 100644 --- a/lib/components/fabro-agent/src/profiles/claude5.rs +++ b/lib/components/fabro-agent/src/profiles/claude5.rs @@ -8,16 +8,14 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, claude5_tools}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps, claude5_tools}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::subagent::{SessionFactory, SubAgentSupervisor}; -use crate::todo_runtime::TodoRuntime; use crate::todo_tools::{ make_task_create_tool, make_task_get_tool, make_task_list_tool, make_task_update_tool, }; use crate::tool_registry::ToolRegistry; -use crate::tools::WebFetchSummarizer; const CORE_PROMPT: &str = include_str!("prompts/claude5.md.j2"); @@ -28,29 +26,15 @@ pub struct Claude5Profile { impl Claude5Profile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::Claude5); - Self::with_native_tools(model, &options, None) + let deps = + ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::Claude5)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - ) -> Self { - Self::with_native_tools_and_todo_runtime( - model, - options, - summarizer, - Arc::new(TodoRuntime::new()), - ) - } - - pub(crate) fn with_native_tools_and_todo_runtime( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - todo_runtime: Arc, - ) -> Self { + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { + let options = &deps.options; + let summarizer = deps.summarizer.clone(); + let todo_runtime = Arc::clone(&deps.todo_runtime); let mut registry = ToolRegistry::with_vocabulary(ToolVocabulary::Claude5); registry.register(claude5_tools::make_read_tool()); registry.register(claude5_tools::make_write_tool()); diff --git a/lib/components/fabro-agent/src/profiles/gemini.rs b/lib/components/fabro-agent/src/profiles/gemini.rs index a3a9fdf5b..c533a85e3 100644 --- a/lib/components/fabro-agent/src/profiles/gemini.rs +++ b/lib/components/fabro-agent/src/profiles/gemini.rs @@ -5,13 +5,13 @@ use fabro_model::{AgentProfileKind, Catalog, ProviderId}; use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::tool_registry::ToolRegistry; use crate::tools::{ - WEB_SEARCH_TOOL_NAME, WebFetchSummarizer, make_edit_file_tool, make_list_dir_tool, - make_read_many_files_tool, register_core_tools, + WEB_SEARCH_TOOL_NAME, make_edit_file_tool, make_list_dir_tool, make_read_many_files_tool, + register_core_tools, }; const CORE_PROMPT: &str = include_str!("prompts/gemini.md.j2"); @@ -23,18 +23,15 @@ pub struct GeminiProfile { impl GeminiProfile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::Gemini); - Self::with_native_tools(model, &options, None) + let deps = + ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::Gemini)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - ) -> Self { + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { let mut registry = ToolRegistry::new(); - register_core_tools(&mut registry, options, summarizer); + register_core_tools(&mut registry, &deps.options, deps.summarizer.clone()); registry.register(make_edit_file_tool()); registry.register(make_read_many_files_tool()); registry.register(make_list_dir_tool()); diff --git a/lib/components/fabro-agent/src/profiles/gpt56.rs b/lib/components/fabro-agent/src/profiles/gpt56.rs index b5a436466..66fa5ddfd 100644 --- a/lib/components/fabro-agent/src/profiles/gpt56.rs +++ b/lib/components/fabro-agent/src/profiles/gpt56.rs @@ -24,7 +24,7 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, FileEditToolKind}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, FileEditToolKind, ProfileDeps}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; @@ -45,11 +45,13 @@ pub struct Gpt56Profile { impl Gpt56Profile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::Gpt56); - Self::with_native_tools(model, &options) + let deps = ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::Gpt56)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools(model: impl Into, options: &NativeToolOptions) -> Self { + /// `deps.summarizer` is ignored: this profile exposes no `web_fetch`. + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { + let options = &deps.options; // The registry carries the vocabulary, so tools registered later -- // subagent tools, skills -- are named consistently too. let mut registry = ToolRegistry::with_vocabulary(ToolVocabulary::Codex); @@ -386,7 +388,8 @@ mod tests { let mut options = NativeToolOptions::for_profile(AgentProfileKind::Gpt56); options.secrets.brave_search_api_key = Some("configured-key".to_string()); - let searching = Gpt56Profile::with_native_tools("gpt-5.6-sol", &options); + let deps = ProfileDeps::standalone(options); + let searching = Gpt56Profile::with_native_tools("gpt-5.6-sol", &deps); assert!(searching.tool_registry().get("web_search").is_some()); assert!(prompt(&searching).contains("web_search")); } diff --git a/lib/components/fabro-agent/src/profiles/kimi.rs b/lib/components/fabro-agent/src/profiles/kimi.rs index f5f8f05fa..1e5bb01de 100644 --- a/lib/components/fabro-agent/src/profiles/kimi.rs +++ b/lib/components/fabro-agent/src/profiles/kimi.rs @@ -6,13 +6,13 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, kimi_tools}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps, kimi_tools}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; use crate::todo_tools::make_todo_list_tool; use crate::tool_registry::ToolRegistry; -use crate::tools::{WebFetchSummarizer, register_discovery_and_web_tools}; +use crate::tools::register_discovery_and_web_tools; const CORE_PROMPT: &str = include_str!("prompts/kimi.md.j2"); @@ -56,15 +56,12 @@ pub struct KimiProfile { impl KimiProfile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::Kimi); - Self::with_native_tools(model, &options, None) + let deps = ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::Kimi)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - ) -> Self { + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { + let options = &deps.options; // The registry carries the vocabulary, so tools registered later // (subagent tools, skills) are renamed too. let mut registry = ToolRegistry::with_vocabulary(ToolVocabulary::KimiCode); @@ -72,7 +69,7 @@ impl KimiProfile { // Glob and the web tools have the same contract in both vocabularies. // The remaining Kimi tools use adapters for their different schemas, // while reusing shared execution helpers where their behavior agrees. - register_discovery_and_web_tools(&mut registry, options, summarizer); + register_discovery_and_web_tools(&mut registry, options, deps.summarizer.clone()); registry.register(kimi_tools::make_kimi_read_tool()); registry.register(kimi_tools::make_kimi_write_tool()); registry.register(kimi_tools::make_kimi_edit_tool(EDIT_FILE_DESCRIPTION)); diff --git a/lib/components/fabro-agent/src/profiles/mod.rs b/lib/components/fabro-agent/src/profiles/mod.rs index 798e39661..23168f13c 100644 --- a/lib/components/fabro-agent/src/profiles/mod.rs +++ b/lib/components/fabro-agent/src/profiles/mod.rs @@ -46,6 +46,32 @@ pub struct AgentProfileBuilder { todo_runtime: Arc, } +/// Everything a profile constructor needs from the builder. +/// +/// Bundled rather than passed positionally so that adding a dependency does +/// not mean editing every profile's signature -- and, more importantly, so a +/// dependency cannot reach some profiles and silently miss others. The shared +/// `todo_runtime` is exactly that case: task tools scope their list by +/// `root_session_id`, so a root and its children address one logical list and +/// must resolve it through one runtime. +pub(crate) struct ProfileDeps { + pub options: NativeToolOptions, + pub summarizer: Option, + pub todo_runtime: Arc, +} + +impl ProfileDeps { + /// Standalone defaults, for `Profile::new` and tests. A profile built this + /// way owns its runtime because it has no children to share one with. + pub(crate) fn standalone(options: NativeToolOptions) -> Self { + Self { + options, + summarizer: None, + todo_runtime: Arc::new(TodoRuntime::new()), + } + } +} + impl AgentProfileBuilder { #[must_use] pub fn new( @@ -84,44 +110,42 @@ impl AgentProfileBuilder { #[must_use] pub fn build(&self) -> Box { let model = self.model.as_str(); - let options = &self.native_tool_options; - let summarizer = if self.profile_kind == AgentProfileKind::Gpt56 { - None - } else { - self.summarizer.clone() + let deps = ProfileDeps { + options: self.native_tool_options.clone(), + summarizer: if self.profile_kind == AgentProfileKind::Gpt56 { + None + } else { + self.summarizer.clone() + }, + todo_runtime: Arc::clone(&self.todo_runtime), }; match self.profile_kind { AgentProfileKind::OpenAi => Box::new( - OpenAiProfile::with_native_tools(model, options, summarizer) + OpenAiProfile::with_native_tools(model, &deps) .with_route(self.provider_id.clone(), Arc::clone(&self.catalog)), ), AgentProfileKind::Gemini => Box::new( - GeminiProfile::with_native_tools(model, options, summarizer) + GeminiProfile::with_native_tools(model, &deps) .with_provider_id(self.provider_id.clone()) .with_catalog(Arc::clone(&self.catalog)), ), AgentProfileKind::Anthropic => Box::new( - AnthropicProfile::with_native_tools(model, options, summarizer) + AnthropicProfile::with_native_tools(model, &deps) .with_provider_id(self.provider_id.clone()) .with_catalog(Arc::clone(&self.catalog)), ), AgentProfileKind::Claude5 => Box::new( - Claude5Profile::with_native_tools_and_todo_runtime( - model, - options, - summarizer, - Arc::clone(&self.todo_runtime), - ) - .with_provider_id(self.provider_id.clone()) - .with_catalog(Arc::clone(&self.catalog)), + Claude5Profile::with_native_tools(model, &deps) + .with_provider_id(self.provider_id.clone()) + .with_catalog(Arc::clone(&self.catalog)), ), AgentProfileKind::Kimi => Box::new( - KimiProfile::with_native_tools(model, options, summarizer) + KimiProfile::with_native_tools(model, &deps) .with_provider_id(self.provider_id.clone()) .with_catalog(Arc::clone(&self.catalog)), ), AgentProfileKind::Gpt56 => Box::new( - Gpt56Profile::with_native_tools(model, options) + Gpt56Profile::with_native_tools(model, &deps) .with_route(self.provider_id.clone(), Arc::clone(&self.catalog)), ), } @@ -422,7 +446,8 @@ mod tests { fn anthropic_profile(has_web_search: bool, has_subagents: bool) -> AnthropicProfile { let options = native_tool_options(AgentProfileKind::Anthropic, has_web_search); - let mut profile = AnthropicProfile::with_native_tools("claude-haiku-4-5", &options, None); + let deps = ProfileDeps::standalone(options); + let mut profile = AnthropicProfile::with_native_tools("claude-haiku-4-5", &deps); if has_subagents { register_test_subagent_tools(&mut profile); } @@ -435,7 +460,8 @@ mod tests { has_question: bool, ) -> Claude5Profile { let options = native_tool_options(AgentProfileKind::Claude5, has_web_search); - let mut profile = Claude5Profile::with_native_tools("claude-sonnet-5", &options, None); + let deps = ProfileDeps::standalone(options); + let mut profile = Claude5Profile::with_native_tools("claude-sonnet-5", &deps); if has_subagents { register_test_subagent_tools(&mut profile); } @@ -450,26 +476,30 @@ mod tests { fn gemini_profile(has_web_search: bool) -> GeminiProfile { let options = native_tool_options(AgentProfileKind::Gemini, has_web_search); - GeminiProfile::with_native_tools("gemini-3-flash-preview", &options, None) + let deps = ProfileDeps::standalone(options); + GeminiProfile::with_native_tools("gemini-3-flash-preview", &deps) } fn openai_apply_patch_profile(has_web_search: bool) -> OpenAiProfile { let options = native_tool_options(AgentProfileKind::OpenAi, has_web_search); - OpenAiProfile::with_native_tools("gpt-5.4-mini", &options, None) + let deps = ProfileDeps::standalone(options); + OpenAiProfile::with_native_tools("gpt-5.4-mini", &deps) } fn gpt56_profile(has_web_search: bool) -> Gpt56Profile { let options = native_tool_options(AgentProfileKind::Gpt56, has_web_search); - Gpt56Profile::with_native_tools("gpt-5.6-sol", &options) + let deps = ProfileDeps::standalone(options); + Gpt56Profile::with_native_tools("gpt-5.6-sol", &deps) } /// GPT-5.6 through an OpenAI-compatible gateway, where `apply_patch` /// cannot be carried and `edit_file` takes its place. fn gpt56_edit_file_profile(has_web_search: bool) -> Gpt56Profile { let options = native_tool_options(AgentProfileKind::Gpt56, has_web_search); + let deps = ProfileDeps::standalone(options); let overrides: LlmCatalogSettings = toml::from_str("[providers.openrouter]\nenabled = true\n").unwrap(); - Gpt56Profile::with_native_tools("gpt-5.6-sol", &options).with_route( + Gpt56Profile::with_native_tools("gpt-5.6-sol", &deps).with_route( ProviderId::new("openrouter"), Arc::new(Catalog::from_builtin_with_overrides(&overrides).unwrap()), ) @@ -477,7 +507,8 @@ mod tests { fn openai_edit_file_profile(has_web_search: bool) -> OpenAiProfile { let options = native_tool_options(AgentProfileKind::OpenAi, has_web_search); - OpenAiProfile::with_native_tools("kimi-k2.5", &options, None).with_route( + let deps = ProfileDeps::standalone(options); + OpenAiProfile::with_native_tools("kimi-k2.5", &deps).with_route( ProviderId::new("kimi"), Arc::new(Catalog::from_builtin().unwrap()), ) @@ -681,37 +712,37 @@ mod tests { } } - #[tokio::test] - async fn claude5_builder_shares_tasks_across_root_and_child_profiles() { + /// Task tools scope their list by `root_session_id`, so a root session and + /// every child it spawns address one logical list. `build()` runs once per + /// session, so the runtime behind that list has to come from the builder -- + /// a per-profile runtime gives each session its own projection and its own + /// ID counter, and the two sessions then collide on `#1` in the merged + /// projection while neither can see the other's tasks. + async fn assert_builder_shares_tasks_across_root_and_child( + profile_kind: AgentProfileKind, + model: &str, + ) { let builder = AgentProfileBuilder::new( - AgentProfileKind::Claude5, + profile_kind, ProviderId::anthropic(), - "claude-sonnet-5", + model, Arc::new(Catalog::from_builtin().unwrap()), ); let root = builder.build(); let child = builder.build(); - let root_create = Arc::clone( - &root - .tool_registry() - .get("TaskCreate") - .expect("root should expose TaskCreate") - .executor, - ); - let child_create = Arc::clone( - &child - .tool_registry() - .get("TaskCreate") - .expect("child should expose TaskCreate") - .executor, - ); - let child_list = Arc::clone( - &child - .tool_registry() - .get("TaskList") - .expect("child should expose TaskList") - .executor, - ); + let executor = |profile: &dyn AgentProfile, name: &str| { + Arc::clone( + &profile + .tool_registry() + .get(name) + .unwrap_or_else(|| panic!("{profile_kind} should expose {name}")) + .executor, + ) + }; + let root_create = executor(root.as_ref(), "TaskCreate"); + let child_create = executor(child.as_ref(), "TaskCreate"); + let child_list = executor(child.as_ref(), "TaskList"); + let env: Arc = Arc::new(MockSandbox::default()); let context = |session_id: &str| ToolContext { env: Arc::clone(&env), @@ -743,6 +774,24 @@ mod tests { assert!(tasks.contains("#2 [pending] Child task"), "{tasks}"); } + #[tokio::test] + async fn claude5_builder_shares_tasks_across_root_and_child_profiles() { + assert_builder_shares_tasks_across_root_and_child( + AgentProfileKind::Claude5, + "claude-sonnet-5", + ) + .await; + } + + #[tokio::test] + async fn anthropic_builder_shares_tasks_across_root_and_child_profiles() { + assert_builder_shares_tasks_across_root_and_child( + AgentProfileKind::Anthropic, + "claude-haiku-4-5", + ) + .await; + } + #[test] fn profile_builder_selects_a_codec_compatible_gpt56_editor() { let overrides: LlmCatalogSettings = diff --git a/lib/components/fabro-agent/src/profiles/openai.rs b/lib/components/fabro-agent/src/profiles/openai.rs index 8eeafdfce..2f5bbf4df 100644 --- a/lib/components/fabro-agent/src/profiles/openai.rs +++ b/lib/components/fabro-agent/src/profiles/openai.rs @@ -6,13 +6,13 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::apply_patch; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt}; +use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; use crate::todo_tools::make_update_plan_tool; use crate::tool_registry::ToolRegistry; -use crate::tools::{self, WebFetchSummarizer, register_core_tools}; +use crate::tools::{self, register_core_tools}; const CORE_PROMPT: &str = include_str!("prompts/openai.md.j2"); @@ -23,18 +23,15 @@ pub struct OpenAiProfile { impl OpenAiProfile { #[must_use] pub fn new(model: impl Into) -> Self { - let options = NativeToolOptions::for_profile(AgentProfileKind::OpenAi); - Self::with_native_tools(model, &options, None) + let deps = + ProfileDeps::standalone(NativeToolOptions::for_profile(AgentProfileKind::OpenAi)); + Self::with_native_tools(model, &deps) } - pub(crate) fn with_native_tools( - model: impl Into, - options: &NativeToolOptions, - summarizer: Option, - ) -> Self { + pub(crate) fn with_native_tools(model: impl Into, deps: &ProfileDeps) -> Self { let mut registry = ToolRegistry::new(); - register_core_tools(&mut registry, options, summarizer); + register_core_tools(&mut registry, &deps.options, deps.summarizer.clone()); registry.register(apply_patch::make_apply_patch_tool()); // Codex-compatible `update_plan` is OpenAI-only. let todo_runtime = Arc::new(TodoRuntime::new()); From 3c755a7d4e7244ea61ecc6bc6221543698f4424c Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 26 Jul 2026 07:57:42 -0400 Subject: [PATCH 07/76] refactor(agent): give Claude 5 subagent tools fabro canonical names `NativeTool` documents itself as "an identity, not a name" whose canonical form is fabro's own vocabulary, with harness names layered on as aliases: `to_string = "read_file", serialize = "Read"`. The four Claude 5 subagent tools inverted that. `ClaudeAgent` declared `to_string = "Agent"`, making the Anthropic wire name the identity and leaving `name(ToolVocabulary::Fabro)` returning `"Agent"` -- and pairing a provider-specific variant name with a generic wire name. It also meant the `Claude5` arm listed none of them: they fell through to `canonical_name()` and were correct only by accident. Rename to `BackgroundAgent` / `AgentOutput` / `StopAgent` / `MessageAgent` with fabro canonical names, keep the harness names as `serialize` aliases so `from_any_name` still resolves them, and name them explicitly in the `Claude5` vocabulary arm. Also map `Grep`/`Glob` there: that arm describes the vocabulary rather than the profile's registry, and if either were ever registered it would otherwise reach the harness lowercased. Records why these are separate identities from `spawn_agent`/`wait`/`close_agent`/`send_input` rather than aliases of them, since the capabilities genuinely differ. Co-Authored-By: Claude Opus 5 (1M context) --- lib/components/fabro-agent/src/native_tool.rs | 72 +++++++++++++++---- .../fabro-agent/src/profiles/claude5.rs | 2 +- .../fabro-agent/src/profiles/claude5_tools.rs | 8 +-- 3 files changed, 64 insertions(+), 18 deletions(-) diff --git a/lib/components/fabro-agent/src/native_tool.rs b/lib/components/fabro-agent/src/native_tool.rs index 83b1890ff..84061fbc8 100644 --- a/lib/components/fabro-agent/src/native_tool.rs +++ b/lib/components/fabro-agent/src/native_tool.rs @@ -73,14 +73,21 @@ pub enum NativeTool { Wait, #[strum(to_string = "close_agent")] CloseAgent, - #[strum(to_string = "Agent")] - ClaudeAgent, - #[strum(to_string = "TaskOutput")] - TaskOutput, - #[strum(to_string = "TaskStop")] - TaskStop, - #[strum(to_string = "SendMessage")] - SendMessage, + // Claude 5 drives one background agent through four tools, where fabro's + // own vocabulary uses `spawn_agent`/`wait`/`close_agent`/`send_input`. + // They are separate identities rather than aliases of those because the + // capabilities differ: `Agent` runs in the background or inline depending + // on `run_in_background`, and `TaskOutput` both polls and waits. Mapping + // them onto the fabro four would promise semantics those tools do not + // have -- the same reason Kimi Code's `Agent` is deliberately unmapped. + #[strum(to_string = "background_agent", serialize = "Agent")] + BackgroundAgent, + #[strum(to_string = "agent_output", serialize = "TaskOutput")] + AgentOutput, + #[strum(to_string = "stop_agent", serialize = "TaskStop")] + StopAgent, + #[strum(to_string = "message_agent", serialize = "SendMessage")] + MessageAgent, #[strum(to_string = "use_skill", serialize = "Skill")] UseSkill, #[strum(to_string = "update_plan")] @@ -135,9 +142,18 @@ impl NativeTool { Self::WriteFile => "Write", Self::EditFile => "Edit", Self::Shell => "Bash", + // Named for completeness: this arm describes the vocabulary, + // not the profile's registry, and the Claude 5 profile + // deliberately registers neither. + Self::Grep => "Grep", + Self::Glob => "Glob", Self::WebSearch => "WebSearch", Self::WebFetch => "WebFetch", Self::UseSkill => "Skill", + Self::BackgroundAgent => "Agent", + Self::AgentOutput => "TaskOutput", + Self::StopAgent => "TaskStop", + Self::MessageAgent => "SendMessage", other => other.canonical_name(), }, ToolVocabulary::KimiCode => match self { @@ -205,10 +221,10 @@ impl NativeTool { | Self::SendInput | Self::Wait | Self::CloseAgent - | Self::ClaudeAgent - | Self::TaskOutput - | Self::TaskStop - | Self::SendMessage => Some(AgentToolCategory::Subagent), + | Self::BackgroundAgent + | Self::AgentOutput + | Self::StopAgent + | Self::MessageAgent => Some(AgentToolCategory::Subagent), // Uncategorized today. Giving these a category would change the CLI // permission gate, which is a behavior change rather than a // classification cleanup, so they keep their existing answer. @@ -299,9 +315,39 @@ mod tests { "WebFetch" ); assert_eq!( - NativeTool::ClaudeAgent.name(ToolVocabulary::Claude5), + NativeTool::BackgroundAgent.name(ToolVocabulary::Claude5), "Agent" ); + assert_eq!( + NativeTool::AgentOutput.name(ToolVocabulary::Claude5), + "TaskOutput" + ); + assert_eq!( + NativeTool::StopAgent.name(ToolVocabulary::Claude5), + "TaskStop" + ); + assert_eq!( + NativeTool::MessageAgent.name(ToolVocabulary::Claude5), + "SendMessage" + ); + } + + /// The harness name is how a tool is expressed, not what it is: the + /// identity keeps a fabro name, and the harness name resolves back to it. + #[test] + fn claude5_subagent_tools_keep_fabro_canonical_names() { + for (tool, canonical, claude5) in [ + (NativeTool::BackgroundAgent, "background_agent", "Agent"), + (NativeTool::AgentOutput, "agent_output", "TaskOutput"), + (NativeTool::StopAgent, "stop_agent", "TaskStop"), + (NativeTool::MessageAgent, "message_agent", "SendMessage"), + ] { + assert_eq!(tool.canonical_name(), canonical); + assert_eq!(tool.name(ToolVocabulary::Fabro), canonical); + assert_eq!(tool.name(ToolVocabulary::Claude5), claude5); + assert_eq!(NativeTool::from_any_name(canonical), Some(tool)); + assert_eq!(NativeTool::from_any_name(claude5), Some(tool)); + } } #[test] diff --git a/lib/components/fabro-agent/src/profiles/claude5.rs b/lib/components/fabro-agent/src/profiles/claude5.rs index 43a2db179..875001c17 100644 --- a/lib/components/fabro-agent/src/profiles/claude5.rs +++ b/lib/components/fabro-agent/src/profiles/claude5.rs @@ -123,7 +123,7 @@ impl AgentProfile for Claude5Profile { "has_agent", self.base .registry - .get_native(NativeTool::ClaudeAgent) + .get_native(NativeTool::BackgroundAgent) .is_some(), ) .with_bool( diff --git a/lib/components/fabro-agent/src/profiles/claude5_tools.rs b/lib/components/fabro-agent/src/profiles/claude5_tools.rs index 663161bab..58278fa1c 100644 --- a/lib/components/fabro-agent/src/profiles/claude5_tools.rs +++ b/lib/components/fabro-agent/src/profiles/claude5_tools.rs @@ -187,7 +187,7 @@ pub(crate) fn make_agent_tool( ) -> RegisteredTool { RegisteredTool { definition: definition( - NativeTool::ClaudeAgent, + NativeTool::BackgroundAgent, "Launch a child agent for an independent task. Agents run in the background by \ default and notify the parent when they finish. Set run_in_background to false to \ wait for the result synchronously.", @@ -281,7 +281,7 @@ fn finished_output( pub(crate) fn make_task_output_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { RegisteredTool { definition: definition( - NativeTool::TaskOutput, + NativeTool::AgentOutput, "Get a background agent's current status or wait for its final output. Automatic \ completion notifications make ordinary polling unnecessary.", serde_json::json!({ @@ -368,7 +368,7 @@ pub(crate) fn make_task_output_tool(supervisor: SubAgentSupervisor) -> Registere pub(crate) fn make_task_stop_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { RegisteredTool { definition: definition( - NativeTool::TaskStop, + NativeTool::StopAgent, "Stop a running background agent by task ID.", serde_json::json!({ "type": "object", @@ -401,7 +401,7 @@ pub(crate) fn make_task_stop_tool(supervisor: SubAgentSupervisor) -> RegisteredT pub(crate) fn make_send_message_tool(supervisor: SubAgentSupervisor) -> RegisteredTool { RegisteredTool { definition: definition( - NativeTool::SendMessage, + NativeTool::MessageAgent, "Send additional instructions to a running background agent by its task ID.", serde_json::json!({ "type": "object", From 8b7d07b84be0c040ad1af5913b10c4a2b31dcc59 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 26 Jul 2026 08:00:21 -0400 Subject: [PATCH 08/76] refactor(agent): deduplicate the BaseProfile accessor delegation All six provider profiles embed a `BaseProfile` and hand-wrote the same six delegating accessors -- 24 identical lines each. What actually distinguishes them is `build_system_prompt`, and for Claude 5, `register_subagent_tools`. Replace the copies with one `impl_base_profile_accessors!()` invocation. A macro rather than trait defaults because three implementors have no `BaseProfile` to delegate to -- `TestProfile`, the workflow crate's `ShutdownTestProfile`, and the server's `AskFabroProfile` -- so a default would need a runtime fallback for a case the compiler can already rule out. Those three keep their hand-written accessors and are untouched. Co-Authored-By: Claude Opus 5 (1M context) --- .../fabro-agent/src/profiles/anthropic.rs | 28 ++----------- .../fabro-agent/src/profiles/claude5.rs | 28 ++----------- .../fabro-agent/src/profiles/gemini.rs | 28 ++----------- .../fabro-agent/src/profiles/gpt56.rs | 28 ++----------- .../fabro-agent/src/profiles/kimi.rs | 28 ++----------- .../fabro-agent/src/profiles/mod.rs | 39 +++++++++++++++++++ .../fabro-agent/src/profiles/openai.rs | 28 ++----------- 7 files changed, 63 insertions(+), 144 deletions(-) diff --git a/lib/components/fabro-agent/src/profiles/anthropic.rs b/lib/components/fabro-agent/src/profiles/anthropic.rs index a34358d81..338c141fa 100644 --- a/lib/components/fabro-agent/src/profiles/anthropic.rs +++ b/lib/components/fabro-agent/src/profiles/anthropic.rs @@ -5,7 +5,9 @@ use fabro_model::{AgentProfileKind, Catalog, ProviderId}; use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, ProfileDeps, impl_base_profile_accessors, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_tools::{ @@ -68,29 +70,7 @@ impl AnthropicProfile { } impl AgentProfile for AnthropicProfile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, diff --git a/lib/components/fabro-agent/src/profiles/claude5.rs b/lib/components/fabro-agent/src/profiles/claude5.rs index 875001c17..ffa4b6ea1 100644 --- a/lib/components/fabro-agent/src/profiles/claude5.rs +++ b/lib/components/fabro-agent/src/profiles/claude5.rs @@ -8,7 +8,9 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps, claude5_tools}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, ProfileDeps, claude5_tools, impl_base_profile_accessors, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::subagent::{SessionFactory, SubAgentSupervisor}; @@ -85,29 +87,7 @@ impl Claude5Profile { } impl AgentProfile for Claude5Profile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, diff --git a/lib/components/fabro-agent/src/profiles/gemini.rs b/lib/components/fabro-agent/src/profiles/gemini.rs index c533a85e3..f6b3d494d 100644 --- a/lib/components/fabro-agent/src/profiles/gemini.rs +++ b/lib/components/fabro-agent/src/profiles/gemini.rs @@ -5,7 +5,9 @@ use fabro_model::{AgentProfileKind, Catalog, ProviderId}; use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, ProfileDeps, impl_base_profile_accessors, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::tool_registry::ToolRegistry; @@ -62,29 +64,7 @@ impl GeminiProfile { } impl AgentProfile for GeminiProfile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, diff --git a/lib/components/fabro-agent/src/profiles/gpt56.rs b/lib/components/fabro-agent/src/profiles/gpt56.rs index 66fa5ddfd..d0fb93ed1 100644 --- a/lib/components/fabro-agent/src/profiles/gpt56.rs +++ b/lib/components/fabro-agent/src/profiles/gpt56.rs @@ -24,7 +24,9 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, FileEditToolKind, ProfileDeps}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, FileEditToolKind, ProfileDeps, impl_base_profile_accessors, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; @@ -179,29 +181,7 @@ fn make_shell_command_tool(options: &NativeToolOptions) -> RegisteredTool { } impl AgentProfile for Gpt56Profile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, diff --git a/lib/components/fabro-agent/src/profiles/kimi.rs b/lib/components/fabro-agent/src/profiles/kimi.rs index 1e5bb01de..b4010deb3 100644 --- a/lib/components/fabro-agent/src/profiles/kimi.rs +++ b/lib/components/fabro-agent/src/profiles/kimi.rs @@ -6,7 +6,9 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::config::NativeToolOptions; use crate::native_tool::{NativeTool, ToolVocabulary}; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps, kimi_tools}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, ProfileDeps, impl_base_profile_accessors, kimi_tools, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; @@ -116,29 +118,7 @@ impl KimiProfile { } impl AgentProfile for KimiProfile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, diff --git a/lib/components/fabro-agent/src/profiles/mod.rs b/lib/components/fabro-agent/src/profiles/mod.rs index 23168f13c..01e851302 100644 --- a/lib/components/fabro-agent/src/profiles/mod.rs +++ b/lib/components/fabro-agent/src/profiles/mod.rs @@ -199,6 +199,45 @@ impl FileEditToolKind { } } +/// Implement the [`AgentProfile`](crate::agent_profile::AgentProfile) +/// accessors that just delegate to an embedded [`BaseProfile`] named `base`. +/// +/// Every profile that owns a `BaseProfile` writes the same six methods; what +/// actually distinguishes them is `build_system_prompt` and, for some, +/// `register_subagent_tools`. Types that implement the trait without a +/// `BaseProfile` -- test doubles, and the server's ask-fabro profile -- write +/// the accessors themselves, which is why this is a macro rather than a set of +/// trait defaults: there is no sensible default for a profile that has no base. +macro_rules! impl_base_profile_accessors { + () => { + fn profile_kind(&self) -> ::fabro_model::AgentProfileKind { + self.base.profile_kind + } + + fn provider_id(&self) -> ::fabro_model::ProviderId { + self.base.provider_id.clone() + } + + fn model(&self) -> &str { + &self.base.model + } + + fn catalog(&self) -> Option<&::fabro_model::Catalog> { + self.base.catalog.as_deref() + } + + fn tool_registry(&self) -> &$crate::tool_registry::ToolRegistry { + &self.base.registry + } + + fn tool_registry_mut(&mut self) -> &mut $crate::tool_registry::ToolRegistry { + &mut self.base.registry + } + }; +} + +pub(crate) use impl_base_profile_accessors; + /// Common fields shared by all provider profiles. /// /// Each concrete profile embeds this struct and delegates `profile_kind()`, diff --git a/lib/components/fabro-agent/src/profiles/openai.rs b/lib/components/fabro-agent/src/profiles/openai.rs index 2f5bbf4df..5f1029e87 100644 --- a/lib/components/fabro-agent/src/profiles/openai.rs +++ b/lib/components/fabro-agent/src/profiles/openai.rs @@ -6,7 +6,9 @@ use super::EnvContext; use crate::agent_profile::AgentProfile; use crate::apply_patch; use crate::config::NativeToolOptions; -use crate::profiles::{self, BaseProfile, EmbeddedPrompt, ProfileDeps}; +use crate::profiles::{ + self, BaseProfile, EmbeddedPrompt, ProfileDeps, impl_base_profile_accessors, +}; use crate::sandbox::Sandbox; use crate::skills::Skill; use crate::todo_runtime::TodoRuntime; @@ -59,29 +61,7 @@ impl OpenAiProfile { } impl AgentProfile for OpenAiProfile { - fn profile_kind(&self) -> AgentProfileKind { - self.base.profile_kind - } - - fn provider_id(&self) -> ProviderId { - self.base.provider_id.clone() - } - - fn model(&self) -> &str { - &self.base.model - } - - fn catalog(&self) -> Option<&Catalog> { - self.base.catalog.as_deref() - } - - fn tool_registry(&self) -> &ToolRegistry { - &self.base.registry - } - - fn tool_registry_mut(&mut self) -> &mut ToolRegistry { - &mut self.base.registry - } + impl_base_profile_accessors!(); fn build_system_prompt( &self, From 73051c9a8adad0385bc37dd3069828091298fdbf Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 26 Jul 2026 08:04:24 -0400 Subject: [PATCH 09/76] refactor(agent): share one normalizer between the question tools `Claude5QuestionToolArgs`/`Claude5Question`/`Claude5Option` differed from the Anthropic trio only in required-ness -- `header: String` rather than `Option`, same for each option's `description`. The JSON Schema already enforces that at the model boundary, so the lenient structs deserialize the strict payload unchanged. `normalize_claude5_questions` then reproduced `normalize_anthropic_questions` plus an inlined copy of `options_from_anthropic`, so `option_key`, `display_text`, and `bounded_display_field` were each applied in two places and could drift. Replace both with one normalizer taking a `QuestionLimits`. The genuine Claude 5 deltas -- at most four questions, two to four options, a twelve-character header cap, required header and option descriptions, and no previews on multi-select -- become data rather than a second code path. Two rules serde used to enforce are now the normalizer's: a missing header and a missing option description. Both are still rejected, with a clearer message than serde's "missing field". `multiSelect` now defaults to false instead of being a deserialization error; the schema still marks it required, which is where that contract belongs. Adds tests pinning the strict rules against the shared normalizer, and one asserting the lenient contract still accepts optional headers and descriptions. Co-Authored-By: Claude Opus 5 (1M context) --- .../fabro-agent/src/question_tools.rs | 276 ++++++++++++------ 1 file changed, 183 insertions(+), 93 deletions(-) diff --git a/lib/components/fabro-agent/src/question_tools.rs b/lib/components/fabro-agent/src/question_tools.rs index 3fcc2980f..63eb58522 100644 --- a/lib/components/fabro-agent/src/question_tools.rs +++ b/lib/components/fabro-agent/src/question_tools.rs @@ -2,6 +2,7 @@ use std::collections::BTreeMap; use std::future::Future; +use std::ops::RangeInclusive; use std::sync::Arc; use async_trait::async_trait; @@ -149,27 +150,41 @@ struct AnthropicOption { preview: Option, } -#[derive(Debug, Deserialize)] -struct Claude5QuestionToolArgs { - questions: Vec, +/// Contract rules the JSON Schema cannot express, and which differ between +/// the two harnesses sharing one normalizer. +struct QuestionLimits { + questions: RangeInclusive, + questions_error: &'static str, + /// `None` leaves the option count unbounded. + options: Option>, + options_error: &'static str, + max_header_chars: Option, + /// Claude 5's schema marks `header` and every option `description` + /// required, so both are validated rather than passed through as given. + require_header_and_descriptions: bool, + /// Claude 5 renders multi-select without a preview pane. + allow_preview_with_multi_select: bool, } -#[derive(Debug, Deserialize)] -#[serde(rename_all = "camelCase")] -struct Claude5Question { - question: String, - header: String, - options: Vec, - multi_select: bool, -} +const ANTHROPIC_QUESTION_LIMITS: QuestionLimits = QuestionLimits { + questions: 1..=usize::MAX, + questions_error: "questions must contain at least one question", + options: None, + options_error: "", + max_header_chars: None, + require_header_and_descriptions: false, + allow_preview_with_multi_select: true, +}; -#[derive(Debug, Deserialize)] -struct Claude5Option { - label: String, - description: String, - #[serde(default)] - preview: Option, -} +const CLAUDE5_QUESTION_LIMITS: QuestionLimits = QuestionLimits { + questions: 1..=4, + questions_error: "questions must contain between one and four questions", + options: Some(2..=4), + options_error: "each question must contain between two and four options", + max_header_chars: Some(12), + require_header_and_descriptions: true, + allow_preview_with_multi_select: false, +}; #[must_use] pub fn is_question_tool(name: &str) -> bool { @@ -285,7 +300,8 @@ fn make_anthropic_question_tool() -> RegisteredTool { executor: Arc::new(|args, ctx| { Box::pin(async move { let parsed: AnthropicQuestionToolArgs = parse_tool_args(args)?; - let questions = normalize_anthropic_questions(parsed)?; + let questions = + normalize_anthropic_questions(parsed, &ANTHROPIC_QUESTION_LIMITS)?; let answers = execute_question_tool(ctx, questions).await?; format_anthropic_answers(&answers) }) @@ -360,8 +376,9 @@ fn make_claude5_question_tool() -> RegisteredTool { }, executor: Arc::new(|args, ctx| { Box::pin(async move { - let parsed: Claude5QuestionToolArgs = parse_tool_args(args)?; - let questions = normalize_claude5_questions(parsed)?; + let parsed: AnthropicQuestionToolArgs = parse_tool_args(args)?; + let questions = + normalize_anthropic_questions(parsed, &CLAUDE5_QUESTION_LIMITS)?; let answers = execute_question_tool(ctx, questions).await?; format_anthropic_answers(&answers) }) @@ -426,50 +443,42 @@ fn normalize_openai_questions(args: OpenAiQuestionToolArgs) -> Result Result, String> { - if args.questions.is_empty() { - return Err("questions must contain at least one question".to_string()); - } - args.questions - .into_iter() - .map(|question| { - let original_question = non_empty(&question.question, "question")?; - Ok(AgentQuestion { - original_id: None, - text: display_text(question.header.as_deref(), &question.question), - header: question.header, - original_question, - question_type: if question.multi_select { - QuestionType::MultiSelect - } else { - QuestionType::MultipleChoice - }, - options: options_from_anthropic(question.options), - allow_freeform: true, - }) - }) - .collect() -} - -fn normalize_claude5_questions( - args: Claude5QuestionToolArgs, -) -> Result, String> { - if !(1..=4).contains(&args.questions.len()) { - return Err("questions must contain between one and four questions".to_string()); + if !limits.questions.contains(&args.questions.len()) { + return Err(limits.questions_error.to_string()); } args.questions .into_iter() .map(|question| { let original_question = non_empty(&question.question, "question")?; - let header = non_empty(&question.header, "question header")?; - if header.chars().count() > 12 { - return Err("question header must contain at most 12 characters".to_string()); + let header = if limits.require_header_and_descriptions { + let header = non_empty( + question.header.as_deref().unwrap_or_default(), + "question header", + )?; + if limits + .max_header_chars + .is_some_and(|max| header.chars().count() > max) + { + return Err(format!( + "question header must contain at most {} characters", + limits.max_header_chars.unwrap_or_default() + )); + } + Some(header) + } else { + question.header + }; + + if let Some(bounds) = &limits.options { + if !bounds.contains(&question.options.len()) { + return Err(limits.options_error.to_string()); + } } - if !(2..=4).contains(&question.options.len()) { - return Err("each question must contain between two and four options".to_string()); - } - if question.multi_select + if !limits.allow_preview_with_multi_select + && question.multi_select && question .options .iter() @@ -480,36 +489,25 @@ fn normalize_claude5_questions( ); } - let options = question - .options - .into_iter() - .enumerate() - .map(|(idx, option)| { - Ok(InterviewOption { - key: option_key(idx), - label: non_empty(&option.label, "option label")?, - description: Some(bounded_display_field( - &non_empty(&option.description, "option description")?, - OPTION_DESCRIPTION_MAX_CHARS, - )), - preview: option - .preview - .map(|value| bounded_display_field(&value, OPTION_PREVIEW_MAX_CHARS)), - }) - }) - .collect::, String>>()?; + // The lenient contract renders the question and header exactly as + // supplied; the strict one has already trimmed them. + let text = if limits.require_header_and_descriptions { + display_text(header.as_deref(), &original_question) + } else { + display_text(header.as_deref(), &question.question) + }; Ok(AgentQuestion { original_id: None, - text: display_text(Some(&header), &original_question), - header: Some(header), + text, + header, original_question, question_type: if question.multi_select { QuestionType::MultiSelect } else { QuestionType::MultipleChoice }, - options, + options: options_from_anthropic(question.options, limits)?, allow_freeform: true, }) }) @@ -531,19 +529,34 @@ fn options_from_openai(options: Vec) -> Vec { .collect() } -fn options_from_anthropic(options: Vec) -> Vec { +fn options_from_anthropic( + options: Vec, + limits: &QuestionLimits, +) -> Result, String> { options .into_iter() .enumerate() - .map(|(idx, option)| InterviewOption { - key: option_key(idx), - label: option.label, - description: option - .description - .map(|value| bounded_display_field(&value, OPTION_DESCRIPTION_MAX_CHARS)), - preview: option - .preview - .map(|value| bounded_display_field(&value, OPTION_PREVIEW_MAX_CHARS)), + .map(|(idx, option)| { + let (label, description) = if limits.require_header_and_descriptions { + ( + non_empty(&option.label, "option label")?, + Some(non_empty( + option.description.as_deref().unwrap_or_default(), + "option description", + )?), + ) + } else { + (option.label, option.description) + }; + Ok(InterviewOption { + key: option_key(idx), + label, + description: description + .map(|value| bounded_display_field(&value, OPTION_DESCRIPTION_MAX_CHARS)), + preview: option + .preview + .map(|value| bounded_display_field(&value, OPTION_PREVIEW_MAX_CHARS)), + }) }) .collect() } @@ -696,7 +709,7 @@ mod tests { })) .unwrap(); - let questions = normalize_anthropic_questions(args).unwrap(); + let questions = normalize_anthropic_questions(args, &ANTHROPIC_QUESTION_LIMITS).unwrap(); assert_eq!(questions[0].question_type, QuestionType::MultiSelect); assert_eq!( @@ -794,7 +807,7 @@ mod tests { #[test] fn claude5_question_contract_is_strict_and_preserves_preview() { - let args: Claude5QuestionToolArgs = serde_json::from_value(json!({ + let args: AnthropicQuestionToolArgs = serde_json::from_value(json!({ "questions": [{ "header": "Approach", "question": "Which approach should we use?", @@ -814,7 +827,7 @@ mod tests { })) .unwrap(); - let questions = normalize_claude5_questions(args).unwrap(); + let questions = normalize_anthropic_questions(args, &CLAUDE5_QUESTION_LIMITS).unwrap(); assert_eq!(questions[0].header.as_deref(), Some("Approach")); assert_eq!( @@ -824,9 +837,86 @@ mod tests { assert!(questions[0].allow_freeform); } + /// The Claude 5 payload is deserialized through the lenient struct now, so + /// the rules its own struct used to enforce are the normalizer's job. + #[test] + fn claude5_limits_reject_what_the_lenient_contract_allows() { + let question = |patch: serde_json::Value| { + let mut base = json!({ + "question": "Which approach?", + "header": "Approach", + "multiSelect": false, + "options": [ + {"label": "First", "description": "One"}, + {"label": "Second", "description": "Two"} + ] + }); + let object = base.as_object_mut().unwrap(); + for (key, value) in patch.as_object().unwrap() { + if value.is_null() { + object.remove(key); + } else { + object.insert(key.clone(), value.clone()); + } + } + base + }; + let normalize = |questions: serde_json::Value| { + let args: AnthropicQuestionToolArgs = + serde_json::from_value(json!({"questions": questions})).unwrap(); + normalize_anthropic_questions(args, &CLAUDE5_QUESTION_LIMITS) + }; + + // A missing header and a missing option description used to be caught + // by serde; the normalizer has to reject them now. + assert!(normalize(json!([question(json!({"header": null}))])).is_err()); + assert!( + normalize(json!([question(json!({ + "options": [{"label": "First"}, {"label": "Second"}] + }))])) + .is_err() + ); + + assert!( + normalize(json!([question(json!({"header": "ThirteenChars"}))])).is_err(), + "header longer than 12 characters" + ); + assert!( + normalize(json!([question(json!({ + "options": [{"label": "Only", "description": "One"}] + }))])) + .is_err(), + "fewer than two options" + ); + assert!( + normalize(json!(vec![question(json!({})); 5])).is_err(), + "more than four questions" + ); + + assert!(normalize(json!([question(json!({}))])).is_ok()); + } + + /// The same payloads stay acceptable under the lenient contract, so the + /// shared normalizer has not tightened the Anthropic tool. + #[test] + fn anthropic_limits_still_accept_optional_headers_and_descriptions() { + let args: AnthropicQuestionToolArgs = serde_json::from_value(json!({ + "questions": [{ + "question": "Which approach?", + "options": [{"label": "First"}] + }] + })) + .unwrap(); + + let questions = normalize_anthropic_questions(args, &ANTHROPIC_QUESTION_LIMITS).unwrap(); + assert_eq!(questions.len(), 1); + assert_eq!(questions[0].header, None); + assert_eq!(questions[0].options[0].description, None); + } + #[test] fn claude5_rejects_previews_for_multi_select_questions() { - let args: Claude5QuestionToolArgs = serde_json::from_value(json!({ + let args: AnthropicQuestionToolArgs = serde_json::from_value(json!({ "questions": [{ "header": "Features", "question": "Which features should we enable?", @@ -846,7 +936,7 @@ mod tests { })) .unwrap(); - assert!(normalize_claude5_questions(args).is_err()); + assert!(normalize_anthropic_questions(args, &CLAUDE5_QUESTION_LIMITS).is_err()); } #[tokio::test] From 0b24649e7617d372d1ff1738ee258719fb767240 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 26 Jul 2026 09:25:47 -0400 Subject: [PATCH 10/76] fix(cli): keep offline validation catalog-free --- lib/apps/fabro-cli/src/commands/run/create.rs | 7 +- lib/apps/fabro-cli/src/commands/run/runner.rs | 7 +- lib/apps/fabro-cli/src/commands/validate.rs | 6 +- lib/apps/fabro-cli/tests/it/cmd/create.rs | 46 +++++++ lib/apps/fabro-cli/tests/it/cmd/validate.rs | 33 +++++ .../fabro-mcp-server/src/manifest_builder.rs | 2 +- .../fabro-server/src/manifest_validation.rs | 23 +++- lib/apps/fabro-server/src/run_manifest.rs | 40 +++--- .../fabro-server/src/run_tool_manifest.rs | 11 +- .../fabro-server/src/server/handler/graph.rs | 2 +- .../fabro-server/src/server/handler/runs.rs | 22 ++- .../src/handler/manager_loop.rs | 42 +++--- .../fabro-workflow/src/operations/create.rs | 109 +++++++++++---- .../fabro-workflow/src/operations/mod.rs | 2 +- .../fabro-workflow/src/operations/validate.rs | 73 +++++++--- .../fabro-workflow/src/pipeline/mod.rs | 8 +- .../fabro-workflow/src/pipeline/transform.rs | 128 +++++++++++------- .../fabro-workflow/src/pipeline/types.rs | 34 ++++- .../fabro-workflow/src/pipeline/validate.rs | 47 ++++--- .../fabro-workflow/tests/it/integration.rs | 23 ++-- 20 files changed, 449 insertions(+), 216 deletions(-) diff --git a/lib/apps/fabro-cli/src/commands/run/create.rs b/lib/apps/fabro-cli/src/commands/run/create.rs index ad4360384..159160d7e 100644 --- a/lib/apps/fabro-cli/src/commands/run/create.rs +++ b/lib/apps/fabro-cli/src/commands/run/create.rs @@ -61,11 +61,8 @@ pub(crate) async fn create_run( None }; - let mut validation = manifest_validation::validate_manifest( - &RunLayer::default(), - &built.manifest, - ctx.catalog()?, - )?; + let mut validation = + manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?; manifest_validation::promote_template_undefined_variables_to_errors(&mut validation); let diagnostics = api_diagnostics_to_local(&validation.workflow.diagnostics); if !quiet { diff --git a/lib/apps/fabro-cli/src/commands/run/runner.rs b/lib/apps/fabro-cli/src/commands/run/runner.rs index 4e6b72b8b..4c6fcf578 100644 --- a/lib/apps/fabro-cli/src/commands/run/runner.rs +++ b/lib/apps/fabro-cli/src/commands/run/runner.rs @@ -263,12 +263,7 @@ impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder { cwd: &Path, user_settings_path: &Path, ) -> fabro_tool::ToolResult { - run_tool_manifest::build_run_tool_manifest( - spec, - cwd, - user_settings_path, - Arc::clone(&self.catalog), - ) + run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, &self.catalog) } } diff --git a/lib/apps/fabro-cli/src/commands/validate.rs b/lib/apps/fabro-cli/src/commands/validate.rs index b87460f28..e157b50b3 100644 --- a/lib/apps/fabro-cli/src/commands/validate.rs +++ b/lib/apps/fabro-cli/src/commands/validate.rs @@ -23,11 +23,7 @@ pub(crate) fn run( user_settings_path: Some(active_settings_path(None)), ..Default::default() })?; - let response = manifest_validation::validate_manifest( - &RunLayer::default(), - &built.manifest, - base_ctx.catalog()?, - )?; + let response = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?; let diagnostics = api_diagnostics_to_local(&response.workflow.diagnostics); if base_ctx.json_output() { diff --git a/lib/apps/fabro-cli/tests/it/cmd/create.rs b/lib/apps/fabro-cli/tests/it/cmd/create.rs index 5a9a4402f..462a0996f 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/create.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/create.rs @@ -101,6 +101,52 @@ fn create_uses_explicit_server_target_and_prints_remote_run_id() { assert_eq!(output_stdout(&output).trim(), run_id.as_str()); } +#[test] +fn create_defers_provider_validation_to_the_server() { + let context = test_context!(); + let server = MockServer::start(); + let run_id = unique_run_id(); + let mock = server.mock(|when, then| { + when.method("POST") + .path("/api/v1/runs") + .body_includes(r#"provider=\"server-only\""#); + then.status(201) + .header("Content-Type", "application/json") + .body(run_status_response(run_id.as_str(), "submitted").to_string()); + }); + let workflow_path = context.temp_dir.join("server-model.fabro"); + context.write_temp( + "server-model.fabro", + r#"digraph ServerModel { + graph [goal="Use a server-owned model"] + start [shape=Mdiamond] + work [prompt="Do work", model="private-model", provider="server-only"] + exit [shape=Msquare] + start -> work -> exit + }"#, + ); + + let output = context + .create_cmd() + .args([ + "--server", + &format!("{}/api/v1", server.base_url()), + "--dry-run", + workflow_path.to_str().unwrap(), + ]) + .output() + .expect("command should execute"); + + assert!( + output.status.success(), + "local validation should not reject a server-owned provider\nstdout:\n{}\nstderr:\n{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + mock.assert(); + assert_eq!(output_stdout(&output).trim(), run_id.as_str()); +} + #[test] fn create_uses_configured_server_target_without_server_flag() { let context = test_context!(); diff --git a/lib/apps/fabro-cli/tests/it/cmd/validate.rs b/lib/apps/fabro-cli/tests/it/cmd/validate.rs index 190c2817b..2bbec8c76 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/validate.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/validate.rs @@ -69,6 +69,39 @@ fn simple_does_not_connect_to_configured_server() { ); } +#[test] +#[expect( + clippy::disallowed_methods, + reason = "sync CLI test writes one workflow fixture before spawning the subprocess" +)] +fn server_owned_provider_is_not_rejected_by_offline_validation() { + let cli = LightweightCli::new(); + let workflow = cli.home().join("server-model.fabro"); + std::fs::write( + &workflow, + r#"digraph ServerModel { + graph [goal="Use a server-owned model"] + start [shape=Mdiamond] + work [prompt="Do work", model="private-model", provider="server-only"] + exit [shape=Msquare] + start -> work -> exit + }"#, + ) + .expect("workflow fixture should be written"); + let mut cmd = cli.command(); + cmd.env("FABRO_SERVER", "http://127.0.0.1:9") + .arg("validate") + .arg(&workflow); + + let output = cmd.output().expect("validate should execute"); + assert!( + output.status.success(), + "offline validation should leave provider availability to the server\nstdout:\n{}\nstderr:\n{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr), + ); +} + #[test] fn branching() { let context = test_context!(); diff --git a/lib/apps/fabro-mcp-server/src/manifest_builder.rs b/lib/apps/fabro-mcp-server/src/manifest_builder.rs index 53e899f1e..8c1cb8125 100644 --- a/lib/apps/fabro-mcp-server/src/manifest_builder.rs +++ b/lib/apps/fabro-mcp-server/src/manifest_builder.rs @@ -32,5 +32,5 @@ fn build_mcp_run_manifest( Catalog::from_builtin_with_overrides(&llm_catalog_settings) .map_err(|err| ToolError::message(err.to_string()))?, ); - run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, catalog) + run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, &catalog) } diff --git a/lib/apps/fabro-server/src/manifest_validation.rs b/lib/apps/fabro-server/src/manifest_validation.rs index baca3f014..3016fde65 100644 --- a/lib/apps/fabro-server/src/manifest_validation.rs +++ b/lib/apps/fabro-server/src/manifest_validation.rs @@ -12,21 +12,34 @@ use crate::run_manifest; pub fn validate_manifest( manifest_run_defaults: &RunLayer, manifest: &types::RunManifest, - catalog: Arc, ) -> Result { validate_manifest_with_environment_defaults( manifest_run_defaults, &fabro_environment::seeded_catalog_layer(), manifest, - catalog, ) } +pub fn validate_manifest_with_catalog( + manifest_run_defaults: &RunLayer, + manifest: &types::RunManifest, + catalog: &Arc, +) -> Result { + let prepared = run_manifest::prepare_manifest_with_environment_defaults( + manifest_run_defaults, + &fabro_environment::seeded_catalog_layer(), + &HashMap::new(), + manifest, + )?; + let validated = + run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?; + Ok(run_manifest::validate_response(&prepared, &validated)) +} + pub fn validate_manifest_with_environment_defaults( manifest_run_defaults: &RunLayer, manifest_environment_defaults: &MergeMap, manifest: &types::RunManifest, - catalog: Arc, ) -> Result { let prepared = run_manifest::prepare_manifest_with_environment_defaults( manifest_run_defaults, @@ -34,8 +47,8 @@ pub fn validate_manifest_with_environment_defaults( &HashMap::new(), manifest, )?; - let validated = - run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?; + let validated = run_manifest::validate_prepared_manifest_structural(&prepared) + .map_err(anyhow::Error::new)?; Ok(run_manifest::validate_response(&prepared, &validated)) } diff --git a/lib/apps/fabro-server/src/run_manifest.rs b/lib/apps/fabro-server/src/run_manifest.rs index 65cc6d36e..76a81db51 100644 --- a/lib/apps/fabro-server/src/run_manifest.rs +++ b/lib/apps/fabro-server/src/run_manifest.rs @@ -35,7 +35,8 @@ use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSecti use fabro_validate::Severity; use fabro_workflow::Error as WorkflowError; use fabro_workflow::operations::{ - CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_ready_providers, + CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_catalog, + validate_with_ready_providers, }; use fabro_workflow::pipeline::Validated; use fabro_workflow::run_materialization::materialize_run_with_ready_providers; @@ -187,34 +188,40 @@ pub(crate) fn prepare_manifest_with_environment_defaults( pub(crate) fn validate_prepared_manifest( prepared: &PreparedManifest, - catalog: Arc, + catalog: &Arc, ) -> Result { validate_prepared_manifest_with_vars(prepared, catalog, HashMap::new()) } +pub(crate) fn validate_prepared_manifest_structural( + prepared: &PreparedManifest, +) -> Result { + validate(manifest_validate_input(prepared, HashMap::new())) +} + pub(crate) fn validate_prepared_manifest_with_vars( prepared: &PreparedManifest, - catalog: Arc, + catalog: &Arc, vars: HashMap, ) -> Result { - validate(manifest_validate_input(prepared, catalog, vars)) + validate_with_catalog(manifest_validate_input(prepared, vars), catalog) } pub(crate) fn validate_prepared_manifest_for_preflight( prepared: &PreparedManifest, - catalog: Arc, + catalog: &Arc, vars: HashMap, ready_providers: &[ProviderId], ) -> Result { validate_with_ready_providers( - manifest_validate_input(prepared, catalog, vars), + manifest_validate_input(prepared, vars), + catalog, ready_providers, ) } fn manifest_validate_input( prepared: &PreparedManifest, - catalog: Arc, vars: HashMap, ) -> ValidateInput { ValidateInput { @@ -223,7 +230,6 @@ fn manifest_validate_input( vars, cwd: prepared.cwd.clone(), custom_transforms: Vec::new(), - catalog, } } @@ -1548,7 +1554,7 @@ digraph Demo {{ .unwrap(); let validated = validate_prepared_manifest_for_preflight( &prepared, - state.catalog(), + &state.catalog(), HashMap::new(), &ready_providers, ) @@ -1633,7 +1639,7 @@ enabled = {clone_enabled} &manifest, ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); let resolved = materialize_run( prepared.settings.clone(), validated.graph(), @@ -2193,7 +2199,7 @@ name = "Control Plane" &invalid_manifest(), ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); assert!(validated.has_errors()); @@ -2238,7 +2244,7 @@ issues = "read" &manifest, ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); assert!(!validated.has_errors()); let (response, _ok) = resolve_and_run_preflight(state.as_ref(), &prepared, &validated) @@ -2288,7 +2294,7 @@ id = "local" &manifest, ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); assert!(!validated.has_errors()); @@ -2397,7 +2403,7 @@ id = "daytona" &manifest, ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); let (response, _ok) = resolve_and_run_preflight(state.as_ref(), &prepared, &validated) .await @@ -2465,7 +2471,7 @@ digraph Demo { &manifest, ) .unwrap(); - let validated = validate_prepared_manifest(&prepared, test_catalog()).unwrap(); + let validated = validate_prepared_manifest(&prepared, &test_catalog()).unwrap(); let (response, ok) = resolve_and_run_preflight(state.as_ref(), &prepared, &validated) .await @@ -2579,7 +2585,7 @@ digraph Demo { &manifest, ) .unwrap(); - let Err(error) = validate_prepared_manifest(&prepared, test_catalog()) else { + let Err(error) = validate_prepared_manifest(&prepared, &test_catalog()) else { panic!("unknown provider should fail static validation"); }; @@ -2646,7 +2652,7 @@ digraph Demo { assert!(ready_providers.is_empty()); let validated = validate_prepared_manifest_for_preflight( &prepared, - state.catalog(), + &state.catalog(), HashMap::new(), &ready_providers, ) diff --git a/lib/apps/fabro-server/src/run_tool_manifest.rs b/lib/apps/fabro-server/src/run_tool_manifest.rs index 8e38e8317..66204e208 100644 --- a/lib/apps/fabro-server/src/run_tool_manifest.rs +++ b/lib/apps/fabro-server/src/run_tool_manifest.rs @@ -14,7 +14,7 @@ pub fn build_run_tool_manifest( spec: &ValidatedCreateRunSpec, cwd: &Path, user_settings_path: &Path, - catalog: Arc, + catalog: &Arc, ) -> ToolResult { let built = fabro_manifest::build_run_manifest(ManifestBuildInput { workflow: PathBuf::from(&spec.workflow), @@ -29,9 +29,12 @@ pub fn build_run_tool_manifest( }) .map_err(|err| ToolError::from_anyhow(&err))?; - let mut validation = - manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest, catalog) - .map_err(|err| ToolError::from_anyhow(&err))?; + let mut validation = manifest_validation::validate_manifest_with_catalog( + &RunLayer::default(), + &built.manifest, + catalog, + ) + .map_err(|err| ToolError::from_anyhow(&err))?; manifest_validation::promote_template_undefined_variables_to_errors(&mut validation); if !validation.ok { return Err(ToolError::message("workflow manifest validation failed")); diff --git a/lib/apps/fabro-server/src/server/handler/graph.rs b/lib/apps/fabro-server/src/server/handler/graph.rs index 364b19cad..ea95b674a 100644 --- a/lib/apps/fabro-server/src/server/handler/graph.rs +++ b/lib/apps/fabro-server/src/server/handler/graph.rs @@ -52,7 +52,7 @@ async fn render_graph_from_manifest( Ok(prepared) => prepared, Err(err) => return ApiError::bad_request(err.to_string()).into_response(), }; - let validated = match run_manifest::validate_prepared_manifest(&prepared, state.catalog()) { + let validated = match run_manifest::validate_prepared_manifest(&prepared, &state.catalog()) { Ok(validated) => validated, Err(err) => return ApiError::bad_request(err.to_string()).into_response(), }; diff --git a/lib/apps/fabro-server/src/server/handler/runs.rs b/lib/apps/fabro-server/src/server/handler/runs.rs index 0e2c8cbcb..a0cbcbf70 100644 --- a/lib/apps/fabro-server/src/server/handler/runs.rs +++ b/lib/apps/fabro-server/src/server/handler/runs.rs @@ -829,7 +829,7 @@ async fn run_preflight( let (llm_result, ready_providers) = state.resolve_llm_client_with_ready_ids().await; let mut validated = match run_manifest::validate_prepared_manifest_for_preflight( &prepared, - state.catalog(), + &state.catalog(), vars, &ready_providers, ) { @@ -879,17 +879,15 @@ async fn validate_run_manifest( return ApiError::bad_request(format!("Run config variable interpolation failed: {err}")) .into_response(); } - let validated = match run_manifest::validate_prepared_manifest_with_vars( - &prepared, - state.catalog(), - vars, - ) { - Ok(validated) => validated, - Err(WorkflowError::Parse(_)) => { - return ApiError::bad_request("Validation failed").into_response(); - } - Err(err) => return ApiError::bad_request(err.to_string()).into_response(), - }; + let validated = + match run_manifest::validate_prepared_manifest_with_vars(&prepared, &state.catalog(), vars) + { + Ok(validated) => validated, + Err(WorkflowError::Parse(_)) => { + return ApiError::bad_request("Validation failed").into_response(); + } + Err(err) => return ApiError::bad_request(err.to_string()).into_response(), + }; ( StatusCode::OK, Json(run_manifest::validate_response(&prepared, &validated)), diff --git a/lib/components/fabro-workflow/src/handler/manager_loop.rs b/lib/components/fabro-workflow/src/handler/manager_loop.rs index 98804d5e9..ddefe84be 100644 --- a/lib/components/fabro-workflow/src/handler/manager_loop.rs +++ b/lib/components/fabro-workflow/src/handler/manager_loop.rs @@ -16,7 +16,7 @@ use crate::artifact_upload::ArtifactSink; use crate::condition::evaluate_condition; use crate::context::{Context, WorkflowContext, context_diff_public, keys}; use crate::error::Error; -use crate::operations::{ValidateInput, WorkflowInput, validate}; +use crate::operations::{ValidateInput, WorkflowInput, validate_with_catalog}; use crate::outcome::{Outcome, OutcomeExt, StageOutcome}; use crate::pipeline::types::Initialized; use crate::run_options::RunOptions; @@ -65,17 +65,19 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result Result Some(workflow.path.clone()), WorkflowInput::Path(_) | WorkflowInput::DotSource { .. } => None, }; - let mut validated = validate(ValidateInput { - workflow, - settings: WorkflowSettings::default(), - vars: std::collections::HashMap::new(), - cwd, - custom_transforms: Vec::new(), - catalog: Arc::clone(&services.run.catalog), - })?; + let mut validated = validate_with_catalog( + ValidateInput { + workflow, + settings: WorkflowSettings::default(), + vars: std::collections::HashMap::new(), + cwd, + custom_transforms: Vec::new(), + }, + &services.run.catalog, + )?; validated.promote_template_undefined_variables_to_errors(); validated.raise_on_errors()?; let (graph, _, _) = validated.into_parts(); diff --git a/lib/components/fabro-workflow/src/operations/create.rs b/lib/components/fabro-workflow/src/operations/create.rs index 2ec40d443..4fff69898 100644 --- a/lib/components/fabro-workflow/src/operations/create.rs +++ b/lib/components/fabro-workflow/src/operations/create.rs @@ -24,7 +24,9 @@ use crate::error::Error; use crate::event::{Event, append_event, to_run_event_at}; use crate::file_resolver::FileResolver; use crate::pipeline::types::PersistOptions; -use crate::pipeline::{self, Persisted, TransformOptions, Validated}; +use crate::pipeline::{ + self, ModelResolutionOptions, Persisted, TransformOptions, Transformed, Validated, +}; use crate::records::RunSpec; use crate::run_lookup::default_scratch_base; use crate::run_materialization::materialize_run; @@ -340,6 +342,69 @@ pub(super) fn preprocess_and_validate( catalog_fallback: bool, catalog: &Arc, ) -> Result { + let model_resolution = ModelResolutionOptions { + catalog: Arc::clone(catalog), + default_provider, + eligible_providers: eligible_providers.iter().cloned().collect(), + catalog_fallback, + }; + let transformed = preprocess( + dot_source, + source_name, + current_dir, + file_resolver, + custom_transforms, + template_context, + goal_override, + render_mode, + Some(model_resolution), + )?; + Ok(pipeline::validate_with_catalog( + transformed, + catalog.as_ref(), + &[], + )) +} + +pub(super) fn preprocess_and_validate_structural( + dot_source: &str, + source_name: Option, + current_dir: Option, + file_resolver: Option>, + custom_transforms: Vec>, + template_context: TemplateContext, + goal_override: Option<&str>, + render_mode: RenderMode, +) -> Result { + let transformed = preprocess( + dot_source, + source_name, + current_dir, + file_resolver, + custom_transforms, + template_context, + goal_override, + render_mode, + None, + )?; + Ok(pipeline::validate(transformed, &[])) +} + +#[expect( + clippy::too_many_arguments, + reason = "pipeline stages have distinct source, rendering, and model-resolution inputs" +)] +fn preprocess( + dot_source: &str, + source_name: Option, + current_dir: Option, + file_resolver: Option>, + custom_transforms: Vec>, + template_context: TemplateContext, + goal_override: Option<&str>, + render_mode: RenderMode, + model_resolution: Option, +) -> Result { let mut parsed = pipeline::parse(dot_source)?; apply_goal_override(&mut parsed.graph, goal_override); @@ -350,12 +415,9 @@ pub(super) fn preprocess_and_validate( source_name, render_mode, custom_transforms, - catalog: Arc::clone(catalog), - default_provider, - eligible_providers: eligible_providers.iter().cloned().collect(), - catalog_fallback, + model_resolution, })?; - Ok(pipeline::validate(transformed, catalog.as_ref(), &[])) + Ok(transformed) } pub(super) fn template_context( @@ -462,7 +524,7 @@ mod tests { use object_store::memory::InMemory; use super::*; - use crate::operations::{ValidateInput, validate}; + use crate::operations::{ValidateInput, validate, validate_with_catalog}; use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE}; use crate::workflow_bundle::BundledWorkflow; fn memory_store() -> Arc { @@ -553,17 +615,19 @@ reasoning = false } fn validate_dot(dot_source: &str, settings: WorkflowSettings) -> Validated { - validate(ValidateInput { - workflow: WorkflowInput::DotSource { - source: dot_source.to_string(), - base_dir: None, + validate_with_catalog( + ValidateInput { + workflow: WorkflowInput::DotSource { + source: dot_source.to_string(), + base_dir: None, + }, + settings, + vars: HashMap::new(), + cwd: PathBuf::from("."), + custom_transforms: Vec::new(), }, - settings, - vars: HashMap::new(), - cwd: PathBuf::from("."), - custom_transforms: Vec::new(), - catalog: test_catalog(), - }) + &test_catalog(), + ) .unwrap() } @@ -852,7 +916,6 @@ reasoning = false vars: HashMap::new(), cwd: PathBuf::from("."), custom_transforms: Vec::new(), - catalog: test_catalog(), }); assert!(result.is_err()); @@ -901,7 +964,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); let file_missing = validate(ValidateInput { @@ -920,7 +982,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); assert_eq!( @@ -944,7 +1005,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); let file_goal = validate(ValidateInput { @@ -963,7 +1023,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); assert_eq!( @@ -1055,7 +1114,6 @@ reasoning = false vars: HashMap::new(), cwd: PathBuf::from("."), custom_transforms: Vec::new(), - catalog: test_catalog(), }); assert!(result.is_err()); } @@ -1100,7 +1158,6 @@ reasoning = false vars: HashMap::new(), cwd: PathBuf::from("."), custom_transforms: vec![Box::new(TagTransform)], - catalog: test_catalog(), }) .unwrap(); validated.raise_on_errors().unwrap(); @@ -1135,7 +1192,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); validated.raise_on_errors().unwrap(); @@ -1177,7 +1233,6 @@ reasoning = false vars: HashMap::new(), cwd: dir.path().to_path_buf(), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); @@ -1227,7 +1282,6 @@ reasoning = false vars: HashMap::new(), cwd: PathBuf::from("."), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); @@ -1278,7 +1332,6 @@ reasoning = false vars: HashMap::new(), cwd: PathBuf::from("."), custom_transforms: Vec::new(), - catalog: test_catalog(), }) .unwrap(); diff --git a/lib/components/fabro-workflow/src/operations/mod.rs b/lib/components/fabro-workflow/src/operations/mod.rs index 38b7d9aaa..9523c5ff5 100644 --- a/lib/components/fabro-workflow/src/operations/mod.rs +++ b/lib/components/fabro-workflow/src/operations/mod.rs @@ -22,7 +22,7 @@ pub use rewind::{RewindInput, RewindOutcome, rewind}; pub use source::WorkflowInput; pub use start::{StartServices, Started, start}; pub use timeline::{ForkTarget, RunTimeline, TimelineEntry, build_timeline, timeline}; -pub use validate::{ValidateInput, validate, validate_with_ready_providers}; +pub use validate::{ValidateInput, validate, validate_with_catalog, validate_with_ready_providers}; pub use crate::pipeline::{LlmSpec, SandboxEnvSpec}; pub use crate::transforms::RenderMode; diff --git a/lib/components/fabro-workflow/src/operations/validate.rs b/lib/components/fabro-workflow/src/operations/validate.rs index c2b990f5c..7ac5e4531 100644 --- a/lib/components/fabro-workflow/src/operations/validate.rs +++ b/lib/components/fabro-workflow/src/operations/validate.rs @@ -5,7 +5,9 @@ use std::sync::Arc; use fabro_model::{Catalog, ProviderId}; use fabro_types::WorkflowSettings; -use super::create::{preprocess_and_validate, template_context}; +use super::create::{ + preprocess_and_validate, preprocess_and_validate_structural, template_context, +}; use super::source::{ResolveWorkflowInput, WorkflowInput, resolve_workflow}; use crate::error::Error; use crate::operations::RenderMode; @@ -20,20 +22,50 @@ pub struct ValidateInput { pub vars: HashMap, pub cwd: PathBuf, pub custom_transforms: Vec>, - pub catalog: Arc, } -/// Parse, transform, and validate a DOT source string. +/// Parse, transform, and structurally validate a DOT source string without a +/// model catalog. /// /// Returns `Validated` even when validation produced errors. Call /// `validated.raise_on_errors()` if the caller wants to fail fast. pub fn validate(input: ValidateInput) -> Result { - let eligible_providers = input - .catalog - .all_provider_ids() - .into_iter() - .collect::>(); - validate_with_eligible_providers(input, &eligible_providers, false) + let ValidateInput { + workflow, + settings, + vars, + cwd, + custom_transforms, + } = input; + let resolved = resolve_workflow(ResolveWorkflowInput { + workflow, + settings, + cwd, + }) + .map_err(|err| Error::Parse(err.to_string()))?; + + preprocess_and_validate_structural( + &resolved.raw_source, + resolved + .dot_path + .as_ref() + .map(|path| path.display().to_string()), + resolved.current_dir, + resolved.file_resolver, + custom_transforms, + template_context(Some(&resolved.settings), vars), + resolved.goal_override.as_deref(), + RenderMode::Structural, + ) +} + +/// Parse, transform, and validate a DOT source string against `catalog`. +pub fn validate_with_catalog( + input: ValidateInput, + catalog: &Arc, +) -> Result { + let eligible_providers = catalog.all_provider_ids().into_iter().collect::>(); + validate_with_eligible_providers(input, catalog, &eligible_providers, false) } /// Parse, transform, and validate, resolving models against the ready @@ -41,20 +73,29 @@ pub fn validate(input: ValidateInput) -> Result { /// provider-readiness selection failures. pub fn validate_with_ready_providers( input: ValidateInput, + catalog: &Arc, ready_providers: &[ProviderId], ) -> Result { - validate_with_eligible_providers(input, ready_providers, true) + validate_with_eligible_providers(input, catalog, ready_providers, true) } fn validate_with_eligible_providers( input: ValidateInput, + catalog: &Arc, eligible_providers: &[ProviderId], catalog_fallback: bool, ) -> Result { + let ValidateInput { + workflow, + settings, + vars, + cwd, + custom_transforms, + } = input; let resolved = resolve_workflow(ResolveWorkflowInput { - workflow: input.workflow, - settings: input.settings, - cwd: input.cwd, + workflow, + settings, + cwd, }) .map_err(|err| Error::Parse(err.to_string()))?; @@ -66,8 +107,8 @@ fn validate_with_eligible_providers( .map(|path| path.display().to_string()), resolved.current_dir, resolved.file_resolver, - input.custom_transforms, - template_context(Some(&resolved.settings), input.vars), + custom_transforms, + template_context(Some(&resolved.settings), vars), resolved.goal_override.as_deref(), RenderMode::Structural, resolved @@ -80,6 +121,6 @@ fn validate_with_eligible_providers( .map(fabro_model::ProviderId::new), eligible_providers, catalog_fallback, - &input.catalog, + catalog, ) } diff --git a/lib/components/fabro-workflow/src/pipeline/mod.rs b/lib/components/fabro-workflow/src/pipeline/mod.rs index d0ba5ae1b..c1cc3b4f9 100644 --- a/lib/components/fabro-workflow/src/pipeline/mod.rs +++ b/lib/components/fabro-workflow/src/pipeline/mod.rs @@ -22,8 +22,8 @@ pub use pull_request::{ }; pub use transform::transform; pub use types::{ - Concluded, Executed, FinalizeOptions, Finalized, InitOptions, Initialized, LlmSpec, Parsed, - Persisted, PullRequestOptions, ResumeState, SandboxEnvSpec, TEMPLATE_UNDEFINED_VARIABLE_RULE, - TransformOptions, Transformed, Validated, + Concluded, Executed, FinalizeOptions, Finalized, InitOptions, Initialized, LlmSpec, + ModelResolutionOptions, Parsed, Persisted, PullRequestOptions, ResumeState, SandboxEnvSpec, + TEMPLATE_UNDEFINED_VARIABLE_RULE, TransformOptions, Transformed, Validated, }; -pub use validate::validate; +pub use validate::{validate, validate_with_catalog}; diff --git a/lib/components/fabro-workflow/src/pipeline/transform.rs b/lib/components/fabro-workflow/src/pipeline/transform.rs index b399d6637..d73b26275 100644 --- a/lib/components/fabro-workflow/src/pipeline/transform.rs +++ b/lib/components/fabro-workflow/src/pipeline/transform.rs @@ -63,13 +63,17 @@ pub fn transform(parsed: Parsed, options: &TransformOptions) -> Result TransformOptions { TransformOptions { - current_dir: None, - file_resolver: None, - template_context: fabro_template::TemplateContext::new(), - source_name: None, - render_mode: crate::operations::RenderMode::Strict, - custom_transforms: vec![], - catalog: test_catalog(), - default_provider: None, - eligible_providers: Catalog::builtin().all_provider_ids(), - catalog_fallback: false, + current_dir: None, + file_resolver: None, + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: crate::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: Some(ModelResolutionOptions::new(test_catalog())), } } @@ -177,16 +180,13 @@ mod tests { ) .unwrap(); let transformed = transform(parsed, &TransformOptions { - current_dir: Some(dir.path().to_path_buf()), - file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), - template_context: fabro_template::TemplateContext::new(), - source_name: None, - render_mode: crate::operations::RenderMode::Strict, - custom_transforms: vec![], - catalog: test_catalog(), - default_provider: None, - eligible_providers: Catalog::builtin().all_provider_ids(), - catalog_fallback: false, + current_dir: Some(dir.path().to_path_buf()), + file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: crate::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: Some(ModelResolutionOptions::new(test_catalog())), }) .unwrap(); @@ -227,21 +227,18 @@ mod tests { ) .unwrap(); let transformed = transform(parsed, &TransformOptions { - current_dir: Some(dir.path().to_path_buf()), - file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), - template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from( - [( + current_dir: Some(dir.path().to_path_buf()), + file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), + template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from([ + ( "task".to_string(), toml::Value::String("Launch".to_string()), - )], - )), - source_name: None, - render_mode: crate::operations::RenderMode::Strict, - custom_transforms: vec![], - catalog: test_catalog(), - default_provider: None, - eligible_providers: Catalog::builtin().all_provider_ids(), - catalog_fallback: false, + ), + ])), + source_name: None, + render_mode: crate::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: Some(ModelResolutionOptions::new(test_catalog())), }) .unwrap(); @@ -349,6 +346,38 @@ mod tests { ); } + #[test] + fn structural_transform_preserves_catalog_owned_model_selection() { + let dot = r#"digraph Test { + graph [goal="Test"] + start [shape=Mdiamond] + work [prompt="Do work", model="private-model", provider="server-only"] + exit [shape=Msquare] + start -> work -> exit + }"#; + let parsed = parse(dot).unwrap(); + let transformed = transform(parsed, &TransformOptions { + current_dir: None, + file_resolver: None, + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: crate::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: None, + }) + .unwrap(); + let work = &transformed.graph.nodes["work"]; + + assert_eq!( + work.attrs.get("model").and_then(AttrValue::as_str), + Some("private-model") + ); + assert_eq!( + work.attrs.get("provider").and_then(AttrValue::as_str), + Some("server-only") + ); + } + #[test] fn transform_reports_goal_self_reference_once_across_passes() { // FileInlining renders the goal for prompt context, but TemplateTransform @@ -365,16 +394,13 @@ mod tests { ) .unwrap(); let transformed = transform(parsed, &TransformOptions { - current_dir: Some(dir.path().to_path_buf()), - file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), - template_context: fabro_template::TemplateContext::new(), - source_name: None, - render_mode: crate::operations::RenderMode::Structural, - custom_transforms: vec![], - catalog: test_catalog(), - default_provider: None, - eligible_providers: Catalog::builtin().all_provider_ids(), - catalog_fallback: false, + current_dir: Some(dir.path().to_path_buf()), + file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))), + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: crate::operations::RenderMode::Structural, + custom_transforms: vec![], + model_resolution: Some(ModelResolutionOptions::new(test_catalog())), }) .unwrap(); diff --git a/lib/components/fabro-workflow/src/pipeline/types.rs b/lib/components/fabro-workflow/src/pipeline/types.rs index 425cba9aa..29f5fbf9e 100644 --- a/lib/components/fabro-workflow/src/pipeline/types.rs +++ b/lib/components/fabro-workflow/src/pipeline/types.rs @@ -359,13 +359,20 @@ pub struct Finalized { /// Options for the TRANSFORM phase. pub struct TransformOptions { - pub current_dir: Option, - pub file_resolver: Option>, - pub template_context: TemplateContext, - pub source_name: Option, - pub render_mode: RenderMode, - pub custom_transforms: Vec>, - pub catalog: Arc, + pub current_dir: Option, + pub file_resolver: Option>, + pub template_context: TemplateContext, + pub source_name: Option, + pub render_mode: RenderMode, + pub custom_transforms: Vec>, + /// Catalog-backed model resolution to perform. `None` preserves authored + /// model and provider selectors for catalog-free structural validation. + pub model_resolution: Option, +} + +/// Catalog-backed model resolution options for the TRANSFORM phase. +pub struct ModelResolutionOptions { + pub catalog: Arc, pub default_provider: Option, pub eligible_providers: HashSet, /// Fall back to the full catalog when the eligible providers cannot @@ -373,6 +380,19 @@ pub struct TransformOptions { pub catalog_fallback: bool, } +impl ModelResolutionOptions { + #[must_use] + pub fn new(catalog: Arc) -> Self { + let eligible_providers = catalog.all_provider_ids(); + Self { + catalog, + default_provider: None, + eligible_providers, + catalog_fallback: false, + } + } +} + /// Options for the FINALIZE phase. pub struct FinalizeOptions { pub run_dir: PathBuf, diff --git a/lib/components/fabro-workflow/src/pipeline/validate.rs b/lib/components/fabro-workflow/src/pipeline/validate.rs index f0cd51dbd..00601b4a4 100644 --- a/lib/components/fabro-workflow/src/pipeline/validate.rs +++ b/lib/components/fabro-workflow/src/pipeline/validate.rs @@ -1,15 +1,28 @@ -use fabro_model::Catalog; use fabro_validate::LintRule; use super::types::{Transformed, Validated}; -/// VALIDATE phase: run lint rules against the transformed graph. +/// VALIDATE phase: run catalog-free lint rules against the transformed graph. /// /// **Infallible.** Always returns `Validated` with diagnostics. Caller decides /// whether to fail via `validated.raise_on_errors()`. -pub fn validate( +pub fn validate(transformed: Transformed, extra_rules: &[&dyn LintRule]) -> Validated { + let Transformed { + graph, + source, + mut diagnostics, + } = transformed; + diagnostics.extend(fabro_validate::validate(&graph, extra_rules)); + Validated::new(graph, source, diagnostics) +} + +/// VALIDATE phase: run catalog-free and catalog-backed lint rules. +/// +/// **Infallible.** Always returns `Validated` with diagnostics. Caller decides +/// whether to fail via `validated.raise_on_errors()`. +pub fn validate_with_catalog( transformed: Transformed, - catalog: &Catalog, + catalog: &fabro_model::Catalog, extra_rules: &[&dyn LintRule], ) -> Validated { let Transformed { @@ -27,34 +40,24 @@ pub fn validate( #[cfg(test)] mod tests { - use fabro_model::Catalog; - use super::*; use crate::pipeline::parse::parse; use crate::pipeline::transform; use crate::pipeline::types::TransformOptions; - fn test_catalog() -> std::sync::Arc { - std::sync::Arc::new(Catalog::from_builtin().unwrap()) - } - fn run_pipeline(dot: &str) -> Validated { - let catalog = test_catalog(); let parsed = parse(dot).unwrap(); let transformed = transform::transform(parsed, &TransformOptions { - current_dir: None, - file_resolver: None, - template_context: fabro_template::TemplateContext::new(), - source_name: None, - render_mode: crate::operations::RenderMode::Strict, - custom_transforms: vec![], - catalog: std::sync::Arc::clone(&catalog), - default_provider: None, - eligible_providers: catalog.all_provider_ids(), - catalog_fallback: false, + current_dir: None, + file_resolver: None, + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: crate::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: None, }) .unwrap(); - validate(transformed, catalog.as_ref(), &[]) + validate(transformed, &[]) } #[test] diff --git a/lib/components/fabro-workflow/tests/it/integration.rs b/lib/components/fabro-workflow/tests/it/integration.rs index dc13bcbf8..b07f600fe 100644 --- a/lib/components/fabro-workflow/tests/it/integration.rs +++ b/lib/components/fabro-workflow/tests/it/integration.rs @@ -4853,7 +4853,9 @@ async fn manager_loop_child_workflow_e2e() { #[tokio::test] async fn import_e2e_through_engine() { - use fabro_workflow::pipeline::{TransformOptions, transform, validate}; + use fabro_workflow::pipeline::{ + ModelResolutionOptions, TransformOptions, transform, validate_with_catalog, + }; let dir = tempfile::tempdir().unwrap(); let catalog = std::sync::Arc::new( @@ -4897,21 +4899,18 @@ async fn import_e2e_through_engine() { ) .expect("parse should succeed"); let transformed = transform(parsed, &TransformOptions { - current_dir: Some(dir.path().to_path_buf()), - file_resolver: Some(std::sync::Arc::new( + current_dir: Some(dir.path().to_path_buf()), + file_resolver: Some(std::sync::Arc::new( fabro_workflow::file_resolver::FilesystemFileResolver::new(None), )), - template_context: fabro_template::TemplateContext::new(), - source_name: None, - render_mode: fabro_workflow::operations::RenderMode::Strict, - custom_transforms: vec![], - catalog: std::sync::Arc::clone(&catalog), - default_provider: None, - eligible_providers: catalog.all_provider_ids(), - catalog_fallback: false, + template_context: fabro_template::TemplateContext::new(), + source_name: None, + render_mode: fabro_workflow::operations::RenderMode::Strict, + custom_transforms: vec![], + model_resolution: Some(ModelResolutionOptions::new(std::sync::Arc::clone(&catalog))), }) .unwrap(); - let validated = validate(transformed, catalog.as_ref(), &[]); + let validated = validate_with_catalog(transformed, catalog.as_ref(), &[]); validated .raise_on_errors() .expect("validation should pass after imports expand"); From 1c82bd90086fa31687223b35b829048b59775950 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 11:25:18 -0400 Subject: [PATCH 11/76] fix(workflow): make publish failures terminal --- docs/internal/events.md | 2 + docs/public/api-reference/fabro-api.yaml | 1 + docs/public/integrations/github.mdx | 7 +- .../src/commands/run/run_progress/event.rs | 1 + .../src/commands/run/run_progress/mod.rs | 1 + lib/apps/fabro-cli/tests/it/cmd/pr_view.rs | 1 + lib/apps/fabro-server/src/demo/mod.rs | 1 + lib/apps/fabro-server/src/install.rs | 23 +- lib/apps/fabro-server/src/server.rs | 16 + .../src/server/handler/pull_requests.rs | 15 + lib/apps/fabro-server/src/server/tests.rs | 14 +- lib/components/fabro-github/src/lib.rs | 128 +++++- lib/components/fabro-store/src/run_state.rs | 2 + lib/components/fabro-workflow/README.md | 3 +- lib/components/fabro-workflow/src/error.rs | 82 +++- .../fabro-workflow/src/event/convert.rs | 2 + .../fabro-workflow/src/event/events.rs | 4 + .../fabro-workflow/src/operations/start.rs | 20 +- .../fabro-workflow/src/pipeline/finalize.rs | 344 ++++++++++++++-- .../fabro-workflow/src/pipeline/mod.rs | 10 +- .../fabro-workflow/src/pipeline/publish.rs | 201 +++++++++ .../src/pipeline/pull_request.rs | 384 ++++++------------ .../fabro-workflow/src/pipeline/types.rs | 60 ++- .../fabro-api/tests/status_round_trip.rs | 1 + .../fabro-types/src/run_event/misc.rs | 2 + lib/foundation/fabro-types/src/status.rs | 1 + .../src/models/failure-reason.ts | 1 + 27 files changed, 989 insertions(+), 338 deletions(-) create mode 100644 lib/components/fabro-workflow/src/pipeline/publish.rs diff --git a/docs/internal/events.md b/docs/internal/events.md index 196925e94..63948d376 100644 --- a/docs/internal/events.md +++ b/docs/internal/events.md @@ -2116,6 +2116,7 @@ These legacy events may appear in older run logs. Current CLI backend runs do no "properties": { "pr_url": "https://github.com/org/repo/pull/42", "pr_number": 42, + "head_sha": "d34db33f", "draft": true } } @@ -2125,6 +2126,7 @@ These legacy events may appear in older run logs. Current CLI backend runs do no |----------|------|-------------| | `pr_url` | string | Pull request URL | | `pr_number` | number | Pull request number | +| `head_sha` | string (optional) | Verified commit SHA at the remote PR head; absent on older events | | `draft` | boolean | Whether the PR is a draft | ### `pull_request.linked` diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml index b028d89c0..2c3a0358a 100644 --- a/docs/public/api-reference/fabro-api.yaml +++ b/docs/public/api-reference/fabro-api.yaml @@ -8894,6 +8894,7 @@ components: type: string enum: - workflow_error + - publish_failed - cancelled - approval_denied - terminated diff --git a/docs/public/integrations/github.mdx b/docs/public/integrations/github.mdx index e70fe5428..741c19e47 100644 --- a/docs/public/integrations/github.mdx +++ b/docs/public/integrations/github.mdx @@ -55,6 +55,7 @@ When you choose the GitHub App strategy, the CLI opens GitHub with a pre-filled | Permission | Level | Purpose | |---|---|---| | Contents | Write | Clone repos, push run branches and checkpoints | + | Workflows | Write | Push changes under `.github/workflows/` | | Metadata | Read | Look up repository installation status | | Pull requests | Write | Create and update PRs from workflows | | Checks | Write | Report workflow status on commits | @@ -219,7 +220,7 @@ When a workflow runs in a remote sandbox (Daytona or Docker), Fabro clones the c 2. SSH URLs (e.g. `git@github.com:owner/repo.git`) are converted to HTTPS 3. Fabro signs a short-lived JWT using the App ID and private key (RS256, 10-minute validity) 4. Using the JWT, Fabro looks up the GitHub App installation for the repository (`GET /repos/\{owner\}/\{repo\}/installation`) -5. Fabro requests a scoped Installation Access Token with `contents: write` permission on the specific repository +5. Fabro requests a scoped Installation Access Token with `contents: write` and `workflows: write` permissions on the specific repository 6. The sandbox clones via HTTPS using `x-access-token` as the username and the token as the password For public repositories, the clone works without credentials. The token is still generated because it's needed for pushing checkpoints. @@ -248,7 +249,9 @@ The upper bound on what Fabro will mint is whatever permissions the GitHub App i ### Checkpoint pushing -After each workflow stage, Fabro [checkpoints](/execution/checkpoints) by pushing the run branch and metadata branch to origin. Inside remote sandboxes, the git remote URL is configured with the Installation Access Token for authenticated pushing. +After each workflow stage, Fabro [checkpoints](/execution/checkpoints) by pushing the run branch and metadata branch to origin. Before a successful run becomes terminal, the publish stage pushes the final commit again and treats failure as a run failure. Inside remote sandboxes, the git remote URL is configured with the Installation Access Token for authenticated pushing. + +When pull request creation is enabled, Fabro then checks that GitHub reports the run branch at the exact final commit before opening the PR. A failed final push, branch check, or PR creation marks the run as failed with `publish_failed`; the terminal run event is emitted only after this step finishes. For long-running workflows, Fabro refreshes the token before each push since Installation Access Tokens are short-lived (typically 1 hour). diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs index 1febfc5b7..d53a591ff 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs @@ -614,6 +614,7 @@ mod tests { repo: "widgets".into(), base_branch: "main".into(), head_branch: "fabro/run/42".into(), + head_sha: "final-sha".into(), title: "Ship the server-side PR".into(), draft: true, }; diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs index b18683354..4b7a04cbf 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs @@ -1383,6 +1383,7 @@ mod tests { repo: "fabro".into(), base_branch: "main".into(), head_branch: "fabro/run/42".into(), + head_sha: "final-sha".into(), title: "Ship the change".into(), draft: true, }); diff --git a/lib/apps/fabro-cli/tests/it/cmd/pr_view.rs b/lib/apps/fabro-cli/tests/it/cmd/pr_view.rs index 4c716c8ea..35a68e4d9 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/pr_view.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/pr_view.rs @@ -85,6 +85,7 @@ fn pr_view_reads_pull_request_from_store_without_pull_request_json() { repo: "fabro".to_string(), base_branch: "main".to_string(), head_branch: "fabro/run/demo".to_string(), + head_sha: Some("final-sha".to_string()), title: "Map the constellations".to_string(), draft: false, }), diff --git a/lib/apps/fabro-server/src/demo/mod.rs b/lib/apps/fabro-server/src/demo/mod.rs index b9d51dcef..554b73672 100644 --- a/lib/apps/fabro-server/src/demo/mod.rs +++ b/lib/apps/fabro-server/src/demo/mod.rs @@ -1280,6 +1280,7 @@ mod runs { fn parse_failure_reason(reason: &str) -> Option { match reason { "workflow_error" => Some(FailureReason::WorkflowError), + "publish_failed" => Some(FailureReason::PublishFailed), "cancelled" => Some(FailureReason::Cancelled), "approval_denied" => Some(FailureReason::ApprovalDenied), "terminated" => Some(FailureReason::Terminated), diff --git a/lib/apps/fabro-server/src/install.rs b/lib/apps/fabro-server/src/install.rs index 12bb5fbb7..ded7f9216 100644 --- a/lib/apps/fabro-server/src/install.rs +++ b/lib/apps/fabro-server/src/install.rs @@ -2053,6 +2053,7 @@ fn build_github_app_manifest( "public": false, "default_permissions": { "contents": "write", + "workflows": "write", "metadata": "read", "pull_requests": "write", "checks": "write", @@ -2354,11 +2355,27 @@ mod tests { InstallObjectStoreCredentialMode, InstallObjectStoreInput, InstallObjectStoreProvider, InstallObjectStoreState, InstallSandboxProviderState, InstallSandboxState, InstallTokenQuery, LlmProvidersInput, PendingInstall, ServerConfigInput, ServerSecrets, - classify_object_store_validation_error, detect_canonical_url, install_object_store_lookup, - lock_unpoisoned, post_install_finish, provider_base_url_override, - resolve_install_object_store_state, token_is_valid, write_artifact_store_metadata, + build_github_app_manifest, classify_object_store_validation_error, detect_canonical_url, + install_object_store_lookup, lock_unpoisoned, post_install_finish, + provider_base_url_override, resolve_install_object_store_state, token_is_valid, + write_artifact_store_metadata, }; + #[test] + fn github_app_manifest_allows_workflow_file_writes() { + let manifest = build_github_app_manifest( + "Fabro Test", + "https://fabro.example/setup", + "https://fabro.example/auth/callback/github", + "https://fabro.example/setup", + ); + + assert_eq!( + manifest["default_permissions"]["workflows"], + serde_json::Value::String("write".to_string()) + ); + } + #[test] fn token_validation_accepts_any_matching_source() { let state = InstallAppState::for_test("expected"); diff --git a/lib/apps/fabro-server/src/server.rs b/lib/apps/fabro-server/src/server.rs index 11a9a0948..c3d3113af 100644 --- a/lib/apps/fabro-server/src/server.rs +++ b/lib/apps/fabro-server/src/server.rs @@ -4183,6 +4183,14 @@ async fn execute_run_in_process(state: Arc, run_id: RunId) { reason: FailureReason::Cancelled, }; } + Err(e @ WorkflowError::Publish { .. }) => { + let detail = e.display_with_causes(); + error!(run_id = %run_id, error = %detail, "Run publish failed"); + managed_run.status = RunStatus::Failed { + reason: FailureReason::PublishFailed, + }; + managed_run.error = Some(detail); + } Err(e) => { error!(run_id = %run_id, error = %e, "Run failed"); managed_run.status = RunStatus::Failed { @@ -4197,6 +4205,14 @@ async fn execute_run_in_process(state: Arc, run_id: RunId) { reason: FailureReason::Cancelled, }; } + Err(e @ WorkflowError::Publish { .. }) => { + let detail = e.display_with_causes(); + error!(run_id = %run_id, error = %detail, "Run publish failed"); + managed_run.status = RunStatus::Failed { + reason: FailureReason::PublishFailed, + }; + managed_run.error = Some(detail); + } Err(e) => { error!(run_id = %run_id, error = %e, "Run failed"); managed_run.status = RunStatus::Failed { diff --git a/lib/apps/fabro-server/src/server/handler/pull_requests.rs b/lib/apps/fabro-server/src/server/handler/pull_requests.rs index f826c813f..a30a55c03 100644 --- a/lib/apps/fabro-server/src/server/handler/pull_requests.rs +++ b/lib/apps/fabro-server/src/server/handler/pull_requests.rs @@ -165,6 +165,7 @@ struct RunPrInputs<'a> { goal: &'a str, base_branch: &'a str, run_branch: &'a str, + final_git_sha: &'a str, diff: &'a str, conclusion: &'a fabro_types::Conclusion, normalized_origin: String, @@ -224,6 +225,17 @@ impl<'a> RunPrInputs<'a> { "run_not_finished", ) })?; + let final_git_sha = conclusion + .final_git_commit_sha + .as_deref() + .filter(|sha| !sha.trim().is_empty()) + .ok_or_else(|| { + ApiError::with_code( + StatusCode::BAD_REQUEST, + "Run has no final git commit SHA — the remote branch cannot be verified.", + "missing_final_git_commit", + ) + })?; if !force && !conclusion.status.is_successful() { return Err(ApiError::with_code( StatusCode::BAD_REQUEST, @@ -240,6 +252,7 @@ impl<'a> RunPrInputs<'a> { goal: run_spec.graph.goal(), base_branch, run_branch, + final_git_sha, diff, conclusion, normalized_origin, @@ -323,6 +336,7 @@ async fn create_run_pull_request( origin_url: &inputs.normalized_origin, base_branch: inputs.base_branch, head_branch: inputs.run_branch, + expected_head_sha: inputs.final_git_sha, goal: inputs.goal, diff: inputs.diff, model: &model, @@ -350,6 +364,7 @@ async fn create_run_pull_request( &created_pull_request.link, &created_pull_request.base_branch, &created_pull_request.head_branch, + &created_pull_request.head_sha, &created_pull_request.title, true, ); diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs index 71b4dfebf..73c9fcd34 100644 --- a/lib/apps/fabro-server/src/server/tests.rs +++ b/lib/apps/fabro-server/src/server/tests.rs @@ -4685,6 +4685,7 @@ channel = "#deploys" repo: "fabro".to_string(), base_branch: "main".to_string(), head_branch: "fabro/run/test".to_string(), + head_sha: "final-sha".to_string(), title: "Ship & notify".to_string(), draft: false, }, @@ -6425,6 +6426,7 @@ async fn create_run_with_pull_request_record( repo: "widgets".to_string(), base_branch: "main".to_string(), head_branch: "feature".to_string(), + head_sha: "final-sha".to_string(), title: title.to_string(), draft: false, }, @@ -6519,7 +6521,7 @@ async fn create_completed_run_ready_for_pull_request( status: "succeeded".to_string(), reason: SuccessReason::Completed, total_usd_micros: None, - final_git_commit_sha: None, + final_git_commit_sha: Some("final-sha".to_string()), final_patch: Some(final_patch.to_string()), diff_summary: None, billing: None, @@ -9058,6 +9060,14 @@ async fn get_run_pull_request_returns_stored_github_association_when_github_pr_i #[tokio::test] async fn create_run_pull_request_creates_and_persists_record() { let github = MockServer::start(); + let branch_mock = github.mock(|when, then| { + when.method("GET") + .path("/repos/acme/widgets/branches/fabro/run/42") + .header("authorization", "Bearer ghu_test"); + then.status(200) + .header("content-type", "application/json") + .body(json!({ "commit": { "sha": "final-sha" } }).to_string()); + }); let create_mock = github.mock(|when, then| { when.method("POST") .path("/repos/acme/widgets/pulls") @@ -9155,6 +9165,7 @@ async fn create_run_pull_request_creates_and_persists_record() { assert_eq!(state_body["pull_request"]["repo"], "widgets"); response_mock.assert_async().await; + branch_mock.assert(); create_mock.assert(); } @@ -16477,6 +16488,7 @@ async fn list_runs_includes_live_metadata_from_run_state() { repo: "repo".to_string(), base_branch: "main".to_string(), head_branch: "fabro/run".to_string(), + head_sha: "final-sha".to_string(), title: "Fix board metadata".to_string(), draft: false, }, diff --git a/lib/components/fabro-github/src/lib.rs b/lib/components/fabro-github/src/lib.rs index bac1f5fbc..d150c4759 100644 --- a/lib/components/fabro-github/src/lib.rs +++ b/lib/components/fabro-github/src/lib.rs @@ -587,7 +587,10 @@ async fn mint_installation_token_with_jwt( }) } -/// Request a scoped Installation Access Token with `contents: write`. +/// Request a scoped Installation Access Token for git writes. +/// +/// The `workflows` permission is required when a pushed commit creates or +/// updates files under `.github/workflows/`. pub async fn create_installation_access_token( client: &impl HttpClient, jwt: &str, @@ -601,7 +604,7 @@ pub async fn create_installation_access_token( owner, repo, base_url, - serde_json::json!({ "contents": "write" }), + serde_json::json!({ "contents": "write", "workflows": "write" }), ) .await } @@ -899,6 +902,55 @@ pub async fn branch_exists( branch_exists_with_client(&client, ctx, owner, repo, branch).await } +/// Return the commit SHA at the head of a GitHub branch. +/// +/// Returns `None` when the branch does not exist. +pub async fn branch_head_sha( + ctx: &GitHubContext<'_>, + owner: &str, + repo: &str, + branch: &str, +) -> anyhow::Result> { + #[derive(Deserialize)] + struct BranchResponse { + commit: BranchCommit, + } + + #[derive(Deserialize)] + struct BranchCommit { + sha: String, + } + + let client = ctx.http_client()?; + let token = ctx + .creds + .resolve_bearer_token( + &client, + owner, + repo, + ctx.base_url, + serde_json::json!({ "contents": "read" }), + ) + .await?; + + let url = format!("{}/repos/{owner}/{repo}/branches/{branch}", ctx.base_url); + let auth = format!("Bearer {token}"); + let resp = HttpClient::request(&client, HttpMethod::Get, &url, &github_headers(&auth), None) + .await + .context("Failed to read remote branch head")?; + + match resp.status { + 200 => { + let branch: BranchResponse = resp + .json() + .context("Failed to parse remote branch response")?; + Ok(Some(branch.commit.sha)) + } + 404 => Ok(None), + status => bail!("Unexpected status {status} reading branch '{branch}'"), + } +} + async fn branch_exists_with_client( client: &impl HttpClient, ctx: &GitHubContext<'_>, @@ -1032,26 +1084,47 @@ pub async fn update_app_webhook_config( /// Resolve git clone credentials for a GitHub repository. /// -/// Returns `(username, password)` for authenticated cloning. +/// Returns `(username, password)` for authenticated cloning and pushing. /// Always generates a token regardless of repo visibility, since the token -/// is needed for pushing from the sandbox. +/// is needed for pushing from the sandbox. The token includes `workflows: +/// write` so a run can publish workflow-file changes. pub async fn resolve_clone_credentials( ctx: &GitHubContext<'_>, owner: &str, repo: &str, +) -> anyhow::Result<(Option, Option)> { + match ctx.creds { + GitHubCredentials::Pat(token) => { + Ok((Some("x-access-token".to_string()), Some(token.clone()))) + } + GitHubCredentials::Installation(token) => Ok(( + Some("x-access-token".to_string()), + Some(token.valid_token()?.to_string()), + )), + GitHubCredentials::App(_) => { + let client = ctx.http_client()?; + resolve_clone_credentials_with_client(&client, ctx, owner, repo).await + } + } +} + +async fn resolve_clone_credentials_with_client( + client: &impl HttpClient, + ctx: &GitHubContext<'_>, + owner: &str, + repo: &str, ) -> anyhow::Result<(Option, Option)> { let token = match ctx.creds { GitHubCredentials::Pat(token) => token.clone(), GitHubCredentials::Installation(token) => token.valid_token()?.to_string(), GitHubCredentials::App(_) => { - let client = ctx.http_client()?; ctx.creds .resolve_bearer_token( - &client, + client, owner, repo, ctx.base_url, - serde_json::json!({ "contents": "write" }), + serde_json::json!({ "contents": "write", "workflows": "write" }), ) .await? } @@ -1775,7 +1848,9 @@ mod tests { r#"{"token": "ghs_xxx", "expires_at": "2099-01-01T00:00:00Z"}"#, ) .with_req_header("Authorization", "Bearer test-jwt") - .with_req_body(r#"{"permissions":{"contents":"write"},"repositories":["repo"]}"#); + .with_req_body( + r#"{"permissions":{"contents":"write","workflows":"write"},"repositories":["repo"]}"#, + ); let token = create_installation_access_token(&mock, "test-jwt", "owner", "repo", "") .await @@ -2296,6 +2371,43 @@ mod tests { ); } + #[tokio::test] + async fn resolve_clone_credentials_requests_workflow_write_permission() { + let mock = MockHttpClient::new() + .on( + HttpMethod::Get, + "/repos/owner/repo/installation", + 200, + r#"{"id": 123}"#, + ) + .on( + HttpMethod::Post, + "/app/installations/123/access_tokens", + 201, + r#"{"token": "ghs_xxx", "expires_at": "2099-01-01T00:00:00Z"}"#, + ) + .with_req_body( + r#"{"permissions":{"contents":"write","workflows":"write"},"repositories":["repo"]}"#, + ); + let credentials = GitHubCredentials::App(GitHubAppCredentials { + app_id: "test".to_string(), + private_key_pem: test_rsa_key().to_string(), + slug: None, + }); + let context = GitHubContext::new(&credentials, ""); + let resolved = resolve_clone_credentials_with_client(&mock, &context, "owner", "repo") + .await + .unwrap(); + + assert_eq!( + resolved, + ( + Some("x-access-token".to_string()), + Some("ghs_xxx".to_string()) + ) + ); + } + #[test] fn installation_token_valid_token_rejects_expired_tokens() { let expired = InstallationToken { diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs index 9ab1e2b2d..4f076b41c 100644 --- a/lib/components/fabro-store/src/run_state.rs +++ b/lib/components/fabro-store/src/run_state.rs @@ -3791,6 +3791,7 @@ mod tests { repo: "fabro".to_string(), base_branch: "main".to_string(), head_branch: "fabro/run/demo".to_string(), + head_sha: Some("final-sha".to_string()), title: "Add run PR chip".to_string(), draft: false, }), @@ -3840,6 +3841,7 @@ mod tests { repo: github_pull_request.repo.clone(), base_branch: "main".to_string(), head_branch: "fabro/run/demo".to_string(), + head_sha: Some("final-sha".to_string()), title: "Add run PR chip".to_string(), draft: false, }), diff --git a/lib/components/fabro-workflow/README.md b/lib/components/fabro-workflow/README.md index 21deba55b..007051592 100644 --- a/lib/components/fabro-workflow/README.md +++ b/lib/components/fabro-workflow/README.md @@ -67,7 +67,8 @@ assert_eq!(graph.goal(), "Run tests"); use fabro_workflow::operations::start; use fabro_workflow::pipeline; -// Use `operations::start(...)` for the full initialize -> execute -> finalize flow. +// Use `operations::start(...)` for the full +// initialize -> execute -> conclude -> publish -> finalize flow. // Use `pipeline::initialize(...)` + `pipeline::execute(...)` when you need partial lifecycle control. ``` diff --git a/lib/components/fabro-workflow/src/error.rs b/lib/components/fabro-workflow/src/error.rs index f52079a29..c134a5e70 100644 --- a/lib/components/fabro-workflow/src/error.rs +++ b/lib/components/fabro-workflow/src/error.rs @@ -285,6 +285,15 @@ pub enum Error { source: Option, }, + #[error("Publish error: {message}")] + Publish { + message: String, + failure_class: FailureCategory, + exec_output_tail: Option, + #[source] + source: Option, + }, + #[error("Handler error: {message}")] Handler { message: String, @@ -420,10 +429,49 @@ impl Error { Self::engine_with_source(message, source) } + /// Build an error for the required publish stage. + pub fn publish(message: impl Into) -> Self { + let message = message.into(); + let failure_class = classify_failure_reason(&message); + Self::Publish { + message, + failure_class, + exec_output_tail: None, + source: None, + } + } + + pub fn publish_with_source( + message: impl Into, + source: impl Into, + ) -> Self { + Self::publish_with_source_and_exec_output_tail(message, source, None) + } + + pub fn publish_with_source_and_exec_output_tail( + message: impl Into, + source: impl Into, + exec_output_tail: Option, + ) -> Self { + let message = message.into(); + let source = SharedError::new(source.into()); + let causes = collect_chain(&source); + let rendered = render_with_causes(&message, &causes); + let failure_class = classify_failure_reason(&rendered); + Self::Publish { + message, + failure_class, + exec_output_tail, + source: Some(source), + } + } + #[must_use] pub fn causes(&self) -> Vec { match self { - Self::Engine { source, .. } | Self::Handler { source, .. } => source + Self::Engine { source, .. } + | Self::Publish { source, .. } + | Self::Handler { source, .. } => source .as_ref() .map_or_else(Vec::new, |source| collect_chain(source)), Self::Template { source, .. } => collect_chain(source), @@ -439,15 +487,17 @@ impl Error { /// Whether this error category is retryable (transient) or terminal. /// - /// Retryable: Handler (transient handler failures), Engine (could be - /// transient), Io (network/disk issues are often transient), - /// Llm (delegates to SdkError). Terminal: Parse, Validation, - /// OutputSchemaValidation, Stylesheet (configuration errors), Checkpoint - /// (storage integrity), Cancelled (explicit cancellation). + /// Retryable: Handler and Engine, I/O, LLM errors when the SDK marks them + /// retryable, and Publish errors classified as transient infrastructure + /// failures. Terminal: Parse, Validation, OutputSchemaValidation, + /// Stylesheet, Checkpoint, and Cancelled. #[must_use] pub fn is_retryable(&self) -> bool { match self { Self::Handler { .. } | Self::Engine { .. } | Self::Io(_) => true, + Self::Publish { failure_class, .. } => { + matches!(failure_class, FailureCategory::TransientInfra) + } Self::Llm(sdk_err) => sdk_err.retryable(), Self::Parse(_) | Self::Validation(_) @@ -483,9 +533,9 @@ impl Error { | Self::Unsupported(_) | Self::OutputSchemaValidation(_) => FailureCategory::Deterministic, Self::Precondition(_) | Self::RunNotFound(_) => FailureCategory::Structural, - Self::Handler { failure_class, .. } | Self::Engine { failure_class, .. } => { - *failure_class - } + Self::Handler { failure_class, .. } + | Self::Engine { failure_class, .. } + | Self::Publish { failure_class, .. } => *failure_class, } } @@ -502,13 +552,18 @@ impl Error { #[must_use] pub fn to_failure_detail(&self) -> FailureDetail { let message = match self { - Self::Engine { message, .. } | Self::Handler { message, .. } => message.clone(), + Self::Engine { message, .. } + | Self::Publish { message, .. } + | Self::Handler { message, .. } => message.clone(), _ => self.to_string(), }; let explicit_exec_output_tail = match self { Self::Engine { exec_output_tail, .. } + | Self::Publish { + exec_output_tail, .. + } | Self::Handler { exec_output_tail, .. } => exec_output_tail.clone(), @@ -1974,6 +2029,7 @@ mod tests { }], }, Error::engine("engine err"), + Error::publish("publish err"), Error::handler("handler err"), Error::Llm(SdkError::Network { message: "refused".into(), @@ -2005,6 +2061,12 @@ mod tests { ); } + #[test] + fn publish_error_is_only_retryable_for_transient_failures() { + assert!(Error::publish("connection timed out").is_retryable()); + assert!(!Error::publish("permission denied").is_retryable()); + } + #[test] fn failure_class_stability() { let messages = [ diff --git a/lib/components/fabro-workflow/src/event/convert.rs b/lib/components/fabro-workflow/src/event/convert.rs index 5cef4e581..42a4ffbc4 100644 --- a/lib/components/fabro-workflow/src/event/convert.rs +++ b/lib/components/fabro-workflow/src/event/convert.rs @@ -1311,6 +1311,7 @@ fn event_body_from_event(event: &Event) -> EventBody { repo, base_branch, head_branch, + head_sha, title, draft, } => EventBody::PullRequestCreated(fabro_types::PullRequestCreatedProps { @@ -1320,6 +1321,7 @@ fn event_body_from_event(event: &Event) -> EventBody { repo: repo.clone(), base_branch: base_branch.clone(), head_branch: head_branch.clone(), + head_sha: (!head_sha.is_empty()).then(|| head_sha.clone()), title: title.clone(), draft: *draft, }), diff --git a/lib/components/fabro-workflow/src/event/events.rs b/lib/components/fabro-workflow/src/event/events.rs index 20803b1fa..f512e80c6 100644 --- a/lib/components/fabro-workflow/src/event/events.rs +++ b/lib/components/fabro-workflow/src/event/events.rs @@ -730,6 +730,8 @@ pub enum Event { repo: String, base_branch: String, head_branch: String, + #[serde(default)] + head_sha: String, title: String, draft: bool, }, @@ -769,6 +771,7 @@ impl Event { record: &PullRequestLink, base_branch: &str, head_branch: &str, + head_sha: &str, title: &str, draft: bool, ) -> Self { @@ -779,6 +782,7 @@ impl Event { repo: record.repo.clone(), base_branch: base_branch.to_string(), head_branch: head_branch.to_string(), + head_sha: head_sha.to_string(), title: title.to_string(), draft, } diff --git a/lib/components/fabro-workflow/src/operations/start.rs b/lib/components/fabro-workflow/src/operations/start.rs index 3b560c7cb..58c9e1c20 100644 --- a/lib/components/fabro-workflow/src/operations/start.rs +++ b/lib/components/fabro-workflow/src/operations/start.rs @@ -37,8 +37,8 @@ use crate::event::{ use crate::handler::HandlerRegistry; use crate::outcome::{Outcome, StageOutcome}; use crate::pipeline::{ - self, FinalizeOptions, Finalized, InitOptions, LlmSpec, Persisted, PullRequestOptions, - ResumeState, SandboxEnvSpec, build_conclusion_from_store, classify_engine_result, + self, FinalizeOptions, Finalized, InitOptions, LlmSpec, Persisted, PublishOptions, ResumeState, + SandboxEnvSpec, build_conclusion_from_store, classify_engine_result, }; #[cfg(test)] use crate::records::Checkpoint; @@ -794,7 +794,7 @@ fn runtime_setup_commands( } impl RunSession { - /// Shared engine: initialize, execute, finalize, pull_request. + /// Shared engine: initialize, execute, conclude, publish, finalize. async fn run( self, persisted: Persisted, @@ -921,14 +921,14 @@ impl RunSession { .expect("last_git_sha mutex should not be poisoned: no code panics while holding this lock") .clone(), }; - let pr_opts = PullRequestOptions { + let publish_opts = PublishOptions { pr_config: self.pr_config, github_app: self.pr_github_app, origin_url: self.pr_origin_url, model: self.pr_model, }; - let concluded = match Box::pin(pipeline::finalize(executed, &finalize_opts)).await { + let concluded = match Box::pin(pipeline::conclude(executed, &finalize_opts)).await { Ok(concluded) => concluded, Err(err) => { self.steering_hub.drain_pending_at_run_end(); @@ -936,7 +936,15 @@ impl RunSession { return Err(err); } }; - let finalized = Box::pin(pipeline::pull_request(concluded, &pr_opts)).await; + let published = Box::pin(pipeline::publish(concluded, &publish_opts)).await; + let finalized = match Box::pin(pipeline::finalize(published, &finalize_opts)).await { + Ok(finalized) => finalized, + Err(err) => { + self.steering_hub.drain_pending_at_run_end(); + store_progress_logger.flush().await; + return Err(err); + } + }; // Emit `agent.steer.dropped { reason: run_ended }` for any // unconsumed pending steers on the success path, then flush. The // scopeguard above re-runs as a no-op (drain is idempotent on an diff --git a/lib/components/fabro-workflow/src/pipeline/finalize.rs b/lib/components/fabro-workflow/src/pipeline/finalize.rs index 949578880..673989a5c 100644 --- a/lib/components/fabro-workflow/src/pipeline/finalize.rs +++ b/lib/components/fabro-workflow/src/pipeline/finalize.rs @@ -9,7 +9,7 @@ use fabro_types::{BilledTokenCounts, DiffSummary, EventBody, RunFailure, RunProj use fabro_util::error::collect_causes; use fabro_util::time::elapsed_ms; -use super::types::{Concluded, Executed, FinalizeOptions}; +use super::types::{Concluded, Executed, FinalizeOptions, Finalized, Published}; use crate::error::{Error, run_failure_from_error, run_failure_from_outcome_failure}; use crate::event::{Event, RunNoticeCode, RunNoticeLevel}; use crate::outcome::{Outcome, StageOutcome}; @@ -56,6 +56,15 @@ pub fn classify_engine_result( reason: FailureReason::Cancelled, }, ), + Err(err @ Error::Publish { .. }) => ( + StageOutcome::Failed { + retry_requested: false, + }, + Some(run_failure_from_error(err, FailureReason::PublishFailed)), + RunStatus::Failed { + reason: FailureReason::PublishFailed, + }, + ), Err(err) => ( StageOutcome::Failed { retry_requested: false, @@ -483,6 +492,9 @@ pub(crate) fn build_terminal_event( Err(Error::Cancelled) => { run_failure_from_error(&Error::Cancelled, FailureReason::Cancelled) } + Err(err @ Error::Publish { .. }) => { + run_failure_from_error(err, FailureReason::PublishFailed) + } Err(err) => run_failure_from_error(err, FailureReason::WorkflowError), Ok(outcome) => { if let Some(failure) = outcome.failure.as_ref() { @@ -521,16 +533,13 @@ async fn stop_sandbox_on_terminal( Ok(()) } -/// FINALIZE phase: build conclusion, write the meta branch, emit the terminal -/// `WorkflowRunCompleted`/`WorkflowRunFailed` event. -/// -/// The terminal event is emitted here (not from `on_run_end`) so observers -/// can't act on "done" before the meta branch writes are flushed. +/// CONCLUDE phase: collect the execution result, final commit, and diff. /// /// # Errors /// -/// Returns `Error` if persisting terminal state fails. -pub async fn finalize(executed: Executed, options: &FinalizeOptions) -> Result { +/// Returns `Error` if the run state needed to build the conclusion cannot be +/// collected. +pub async fn conclude(executed: Executed, options: &FinalizeOptions) -> Result { let Executed { graph, outcome, @@ -561,20 +570,78 @@ pub async fn finalize(executed: Executed, options: &FinalizeOptions) -> Result Result { + let Published { + execution_outcome, + publish_outcome, + mut conclusion, + artifact_count, + run_options, + services, + } = published; + + let pushed_branch = publish_outcome + .as_ref() + .ok() + .and_then(|outcome| outcome.pushed_branch()) + .map(str::to_string); + let pr_url = publish_outcome + .as_ref() + .ok() + .and_then(|outcome| outcome.pr_url()) + .map(str::to_string); + let outcome = match (execution_outcome, publish_outcome) { + (Err(error), _) | (Ok(_), Err(error)) => Err(error), + (Ok(outcome), Ok(_)) => Ok(outcome), + }; + + let (final_status, failure, _run_status) = classify_engine_result(&outcome); + conclusion.status = final_status; + conclusion.failure = failure; + + write_finalize_commit(&run_options, &services, &conclusion).await; if services.metadata_runtime.metadata_degraded() { services.emitter.notice( @@ -588,9 +655,9 @@ pub async fn finalize(executed: Executed, options: &FinalizeOptions) -> Result Result Result { + let concluded = conclude(executed, options).await?; + let published = crate::pipeline::publish(concluded, &crate::pipeline::PublishOptions { + pr_config: None, + github_app: None, + origin_url: None, + model: "test-model".to_string(), + }) + .await; + finalize(published, options).await + } + fn test_store() -> Arc { Arc::new(Database::new( Arc::new(InMemory::new()), @@ -869,6 +951,26 @@ mod tests { use crate::test_support::test_usage; + #[test] + fn publish_error_builds_publish_failed_terminal_event() { + let event = build_terminal_event( + &Err(Error::publish("GitHub rejected pull request creation")), + fabro_types::RunTiming::wall_only(10), + 0, + Some("final-sha".to_string()), + Some("diff".to_string()), + None, + None, + ); + + match event { + Event::WorkflowRunFailed { failure, .. } => { + assert_eq!(failure.reason, FailureReason::PublishFailed); + } + other => panic!("expected run failure, got {other:?}"), + } + } + #[test] fn conclusion_stage_order_follows_projection_first_event_order() { let mut projection = test_projection(); @@ -1068,7 +1170,7 @@ mod tests { services, ); - let concluded = finalize(executed, &FinalizeOptions { + let concluded = finalize_executed(executed, &FinalizeOptions { run_dir: run_dir.clone(), run_id: test_run_id(), workflow_name: "test".to_string(), @@ -1249,7 +1351,7 @@ mod tests { services, ); - finalize(executed, &FinalizeOptions { + finalize_executed(executed, &FinalizeOptions { run_dir: repo_dir.path().to_path_buf(), run_id: test_run_id(), workflow_name: "test".to_string(), @@ -1273,6 +1375,196 @@ mod tests { ]); } + #[tokio::test] + async fn configured_run_branch_without_remote_is_not_reported_as_pushed() { + let repo_dir = tempfile::tempdir().unwrap(); + let emitter = Arc::new(Emitter::new(test_run_id())); + let events = record_events(&emitter); + let services = test_services( + RunStoreHandle::local(seeded_run_store().await), + emitter, + Arc::new(MockSandbox::linux()), + Arc::new(RunMetadataRuntime::new()), + None, + ); + let mut run_options = test_run_options(repo_dir.path()); + run_options.git = Some(GitCheckpointOptions { + base_sha: None, + run_branch: Some("fabro/run/test".to_string()), + meta_branch: None, + }); + let executed = test_executed( + Graph::new("test"), + Ok(Outcome::success()), + run_options, + 5, + services, + ); + let options = FinalizeOptions { + run_dir: repo_dir.path().to_path_buf(), + run_id: test_run_id(), + workflow_name: "test".to_string(), + preserve_sandbox: false, + stop_on_terminal: true, + last_git_sha: Some("final-sha".to_string()), + }; + let concluded = conclude(executed, &options).await.unwrap(); + let published = crate::pipeline::publish(concluded, &crate::pipeline::PublishOptions { + pr_config: None, + github_app: None, + origin_url: None, + model: "test-model".to_string(), + }) + .await; + + assert!(matches!( + &published.publish_outcome, + Ok(crate::pipeline::PublishOutcome::NotRequested) + )); + let finalized = finalize(published, &options).await.unwrap(); + + assert!(finalized.outcome.is_ok()); + assert_eq!(finalized.pushed_branch, None); + let events = events.lock().unwrap(); + let names = events.iter().map(RunEvent::event_name).collect::>(); + assert_eq!(names, vec!["run.completed"]); + } + + #[tokio::test] + async fn final_push_failure_becomes_terminal_publish_failure() { + let repo_dir = tempfile::tempdir().unwrap(); + let sandbox = Arc::new(MockSandbox::linux()); + let emitter = Arc::new(Emitter::new(test_run_id())); + let events = record_events(&emitter); + let services = test_services( + RunStoreHandle::local(seeded_run_store().await), + emitter, + sandbox, + Arc::new(RunMetadataRuntime::new()), + None, + ); + let mut run_options = test_run_options(repo_dir.path()); + run_options.git = Some(GitCheckpointOptions { + base_sha: None, + run_branch: Some("fabro/run/test".to_string()), + meta_branch: None, + }); + let executed = test_executed( + Graph::new("test"), + Ok(Outcome::success()), + run_options, + 5, + services, + ); + let options = FinalizeOptions { + run_dir: repo_dir.path().to_path_buf(), + run_id: test_run_id(), + workflow_name: "test".to_string(), + preserve_sandbox: false, + stop_on_terminal: true, + last_git_sha: Some("final-sha".to_string()), + }; + let concluded = conclude(executed, &options).await.unwrap(); + let published = crate::pipeline::publish(concluded, &crate::pipeline::PublishOptions { + pr_config: None, + github_app: None, + origin_url: Some("https://github.com/owner/repo.git".to_string()), + model: "test-model".to_string(), + }) + .await; + + assert!(matches!( + &published.publish_outcome, + Err(Error::Publish { .. }) + )); + let finalized = finalize(published, &options).await.unwrap(); + + assert!(matches!(finalized.outcome, Err(Error::Publish { .. }))); + assert_eq!( + finalized + .conclusion + .failure + .as_ref() + .map(|failure| failure.reason), + Some(FailureReason::PublishFailed) + ); + let events = events.lock().unwrap(); + let names = events.iter().map(RunEvent::event_name).collect::>(); + assert_eq!(names, vec!["git.push", "run.failed"]); + match &events.last().unwrap().body { + EventBody::RunFailed(props) => { + assert_eq!(props.failure.reason, FailureReason::PublishFailed); + } + other => panic!("expected run.failed, got {other:?}"), + } + } + + #[tokio::test] + async fn pull_request_failure_precedes_terminal_publish_failure() { + let repo_dir = tempfile::tempdir().unwrap(); + init_git_repo(repo_dir.path()); + let emitter = Arc::new(Emitter::new(test_run_id())); + let events = record_events(&emitter); + let services = test_services( + RunStoreHandle::local(seeded_run_store().await), + emitter, + Arc::new(fabro_agent::LocalSandbox::new( + repo_dir.path().to_path_buf(), + )), + Arc::new(RunMetadataRuntime::new()), + None, + ); + let mut run_options = test_run_options(repo_dir.path()); + run_options.base_branch = Some("main".to_string()); + run_options.git = Some(GitCheckpointOptions { + base_sha: None, + run_branch: Some("fabro/run/test".to_string()), + meta_branch: None, + }); + let executed = test_executed( + Graph::new("test"), + Ok(Outcome::success()), + run_options, + 5, + services, + ); + let options = FinalizeOptions { + run_dir: repo_dir.path().to_path_buf(), + run_id: test_run_id(), + workflow_name: "test".to_string(), + preserve_sandbox: false, + stop_on_terminal: true, + last_git_sha: Some("final-sha".to_string()), + }; + let mut concluded = conclude(executed, &options).await.unwrap(); + concluded.conclusion.diff.patch = + Some("diff --git a/a b/a\n+published change\n".to_string()); + let published = crate::pipeline::publish(concluded, &crate::pipeline::PublishOptions { + pr_config: Some(fabro_types::settings::run::PullRequestSettings { + enabled: true, + draft: true, + auto_merge: false, + merge_strategy: fabro_types::settings::run::MergeStrategy::Squash, + }), + github_app: None, + origin_url: Some("https://github.com/owner/repo.git".to_string()), + model: "test-model".to_string(), + }) + .await; + let finalized = finalize(published, &options).await.unwrap(); + + assert!(matches!(finalized.outcome, Err(Error::Publish { .. }))); + let events = events.lock().unwrap(); + let names = events.iter().map(RunEvent::event_name).collect::>(); + assert_eq!(names, vec!["git.push", "pull_request.failed", "run.failed"]); + match &events.last().unwrap().body { + EventBody::RunFailed(props) => { + assert_eq!(props.failure.reason, FailureReason::PublishFailed); + } + other => panic!("expected run.failed, got {other:?}"), + } + } + #[tokio::test] async fn finalize_stops_sandbox_on_terminal_without_deleting() { let repo_dir = tempfile::tempdir().unwrap(); @@ -1292,7 +1584,7 @@ mod tests { services, ); - finalize(executed, &FinalizeOptions { + finalize_executed(executed, &FinalizeOptions { run_dir: repo_dir.path().to_path_buf(), run_id: test_run_id(), workflow_name: "test".to_string(), @@ -1326,7 +1618,7 @@ mod tests { services, ); - finalize(executed, &FinalizeOptions { + finalize_executed(executed, &FinalizeOptions { run_dir: repo_dir.path().to_path_buf(), run_id: test_run_id(), workflow_name: "test".to_string(), @@ -1379,7 +1671,7 @@ mod tests { services, ); - finalize(executed, &FinalizeOptions { + finalize_executed(executed, &FinalizeOptions { run_dir: repo.to_path_buf(), run_id: test_run_id(), workflow_name: "test".to_string(), diff --git a/lib/components/fabro-workflow/src/pipeline/mod.rs b/lib/components/fabro-workflow/src/pipeline/mod.rs index d0ba5ae1b..57fd59e15 100644 --- a/lib/components/fabro-workflow/src/pipeline/mod.rs +++ b/lib/components/fabro-workflow/src/pipeline/mod.rs @@ -3,6 +3,7 @@ mod finalize; mod initialize; mod parse; mod persist; +mod publish; mod pull_request; mod transform; pub(crate) mod types; @@ -12,18 +13,19 @@ pub use execute::execute; pub(crate) use finalize::build_conclusion_from_store; #[cfg(any(test, feature = "test-support"))] pub(crate) use finalize::{billing_from_projection, build_terminal_event}; -pub use finalize::{classify_engine_result, finalize, write_finalize_commit}; +pub use finalize::{classify_engine_result, conclude, finalize, write_finalize_commit}; pub use initialize::initialize; pub use parse::parse; pub(crate) use persist::persist; +pub use publish::publish; pub use pull_request::{ AutoMergeOptions, CreatedPullRequest, OpenPullRequestRequest, PrContent, build_pr_content, - maybe_open_pull_request, pull_request, + maybe_open_pull_request, }; pub use transform::transform; pub use types::{ Concluded, Executed, FinalizeOptions, Finalized, InitOptions, Initialized, LlmSpec, Parsed, - Persisted, PullRequestOptions, ResumeState, SandboxEnvSpec, TEMPLATE_UNDEFINED_VARIABLE_RULE, - TransformOptions, Transformed, Validated, + Persisted, PublishOptions, PublishOutcome, Published, ResumeState, SandboxEnvSpec, + TEMPLATE_UNDEFINED_VARIABLE_RULE, TransformOptions, Transformed, Validated, }; pub use validate::validate; diff --git a/lib/components/fabro-workflow/src/pipeline/publish.rs b/lib/components/fabro-workflow/src/pipeline/publish.rs new file mode 100644 index 000000000..d4ba2fd61 --- /dev/null +++ b/lib/components/fabro-workflow/src/pipeline/publish.rs @@ -0,0 +1,201 @@ +use std::sync::Arc; + +use super::pull_request::{AutoMergeOptions, OpenPullRequestRequest, maybe_open_pull_request}; +use super::types::{Concluded, PublishOptions, PublishOutcome, Published}; +use crate::error::Error; +use crate::event::Event; +use crate::outcome::StageOutcome; + +/// PUBLISH phase: push the final run commit and, when configured, open a pull +/// request. +/// +/// Publish is always present in the pipeline. It becomes a no-op when the run +/// did not succeed, is a dry run, or has no remote branch configured. +pub async fn publish(concluded: Concluded, options: &PublishOptions) -> Published { + let publish_outcome = publish_inner(&concluded, options).await; + let Concluded { + outcome, + conclusion, + artifact_count, + graph: _, + run_options, + services, + } = concluded; + + Published { + execution_outcome: outcome, + publish_outcome, + conclusion, + artifact_count, + run_options, + services, + } +} + +async fn publish_inner( + concluded: &Concluded, + options: &PublishOptions, +) -> Result { + let successful_execution = concluded.outcome.as_ref().is_ok_and(|outcome| { + matches!( + outcome.status, + StageOutcome::Succeeded | StageOutcome::PartiallySucceeded + ) + }); + if !successful_execution || concluded.run_options.dry_run_enabled() { + return Ok(PublishOutcome::NotRequested); + } + + let pull_request_requested = options.pr_config.is_some(); + let Some(origin_url) = options + .origin_url + .as_deref() + .filter(|origin| !origin.trim().is_empty()) + else { + if pull_request_requested { + return Err(pull_request_error( + concluded, + "pull request creation requires a GitHub origin URL", + )); + } + return Ok(PublishOutcome::NotRequested); + }; + let Some(run_branch) = concluded.run_options.run_branch() else { + if pull_request_requested { + return Err(pull_request_error( + concluded, + "pull request creation requires a run branch", + )); + } + return Ok(PublishOutcome::NotRequested); + }; + if !concluded.run_options.settings.run.run_branch.push { + if pull_request_requested { + return Err(pull_request_error( + concluded, + "pull request creation requires run branch pushing", + )); + } + return Ok(PublishOutcome::NotRequested); + } + + let final_sha = concluded + .conclusion + .final_git_commit_sha + .as_deref() + .ok_or_else(|| Error::publish("cannot publish a run without a final git commit SHA"))?; + let refspec = format!("refs/heads/{run_branch}:refs/heads/{run_branch}"); + match concluded.services.sandbox.git_push_ref(&refspec).await { + Ok(()) => { + concluded.services.emitter.emit(&Event::GitPush { + branch: run_branch.to_string(), + success: true, + exec_output_tail: None, + }); + } + Err(error) => { + let exec_output_tail = fabro_sandbox::default_redacted_output_tail(&error); + concluded.services.emitter.emit(&Event::GitPush { + branch: run_branch.to_string(), + success: false, + exec_output_tail: exec_output_tail.clone(), + }); + return Err(Error::publish_with_source_and_exec_output_tail( + format!("failed to push final commit {final_sha} to branch '{run_branch}'"), + error, + exec_output_tail, + )); + } + } + + let diff = concluded + .conclusion + .diff + .patch + .as_deref() + .unwrap_or_default(); + let Some(pr_config) = options.pr_config.as_ref() else { + return Ok(PublishOutcome::Published { + pushed_branch: run_branch.to_string(), + pr_url: None, + }); + }; + if diff.trim().is_empty() { + return Ok(PublishOutcome::NoChanges { + pushed_branch: run_branch.to_string(), + }); + } + + let base_branch = concluded + .run_options + .base_branch + .as_deref() + .ok_or_else(|| { + pull_request_error(concluded, "pull request creation requires a base branch") + })?; + let credentials = options.github_app.as_ref().ok_or_else(|| { + pull_request_error( + concluded, + "pull request creation requires GitHub credentials", + ) + })?; + let auto_merge = pr_config.auto_merge.then_some(AutoMergeOptions { + merge_strategy: pr_config.merge_strategy, + }); + let github_base_url = fabro_github::github_api_base_url(); + + let created = maybe_open_pull_request(OpenPullRequestRequest { + github: fabro_github::GitHubContext::new(credentials, &github_base_url), + origin_url, + base_branch, + head_branch: run_branch, + expected_head_sha: final_sha, + goal: concluded.graph.goal(), + diff, + model: &options.model, + draft: pr_config.draft, + auto_merge, + run_store: &concluded.services.run_store, + llm_source: concluded.services.llm_source.as_ref(), + catalog: Arc::clone(&concluded.services.catalog), + conclusion: Some(&concluded.conclusion), + run_state: None, + }) + .await + .map_err(|error| { + concluded.services.emitter.emit(&Event::PullRequestFailed { + error: error.clone(), + }); + Error::publish_with_source("failed to create pull request", anyhow::anyhow!(error)) + })? + .ok_or_else(|| { + pull_request_error( + concluded, + "pull request creation found no changes after the stored diff was checked", + ) + })?; + + concluded + .services + .emitter + .emit(&Event::pull_request_created( + &created.link, + &created.base_branch, + &created.head_branch, + &created.head_sha, + &created.title, + pr_config.draft, + )); + + Ok(PublishOutcome::Published { + pushed_branch: run_branch.to_string(), + pr_url: Some(created.link.html_url()), + }) +} + +fn pull_request_error(concluded: &Concluded, message: &str) -> Error { + concluded.services.emitter.emit(&Event::PullRequestFailed { + error: message.to_string(), + }); + Error::publish(message) +} diff --git a/lib/components/fabro-workflow/src/pipeline/pull_request.rs b/lib/components/fabro-workflow/src/pipeline/pull_request.rs index 0dcc7745c..638c87847 100644 --- a/lib/components/fabro-workflow/src/pipeline/pull_request.rs +++ b/lib/components/fabro-workflow/src/pipeline/pull_request.rs @@ -13,9 +13,7 @@ use fabro_types::settings::run::MergeStrategy; use fabro_util::text::strip_goal_decoration; use tracing::{debug, info, warn}; -use super::types::{Concluded, Finalized, PullRequestOptions}; -use crate::event::{Event, RunNoticeCode, RunNoticeLevel}; -use crate::outcome::{StageOutcome, format_cost as outcome_format_cost}; +use crate::outcome::format_cost as outcome_format_cost; use crate::records::{Conclusion, RunSpec}; use crate::runtime_store::RunStoreHandle; @@ -327,22 +325,6 @@ fn assemble_pr_body( parts.join("\n") } -async fn load_pull_request_diff(run_store: &RunStoreHandle) -> String { - run_store - .state() - .await - .inspect_err(|err| { - tracing::warn!(error = %err, "Failed to load final patch from store for PR"); - }) - .ok() - .and_then(|state| { - state - .conclusion - .and_then(|conclusion| conclusion.diff.patch) - }) - .unwrap_or_default() -} - /// Build complete PR content by combining LLM-generated narrative with /// deterministic fallbacks and programmatic sections. pub async fn build_pr_content( @@ -461,20 +443,23 @@ pub struct AutoMergeOptions { /// Inputs for [`maybe_open_pull_request`]. pub struct OpenPullRequestRequest<'a> { - pub github: github_app::GitHubContext<'a>, - pub origin_url: &'a str, - pub base_branch: &'a str, - pub head_branch: &'a str, - pub goal: &'a str, - pub diff: &'a str, - pub model: &'a str, - pub draft: bool, - pub auto_merge: Option, - pub run_store: &'a RunStoreHandle, - pub llm_source: &'a dyn CredentialSource, - pub catalog: Arc, - pub conclusion: Option<&'a Conclusion>, - pub run_state: Option<&'a RunProjection>, + pub github: github_app::GitHubContext<'a>, + pub origin_url: &'a str, + pub base_branch: &'a str, + pub head_branch: &'a str, + /// Commit that must be visible at the remote branch before the PR is + /// opened. + pub expected_head_sha: &'a str, + pub goal: &'a str, + pub diff: &'a str, + pub model: &'a str, + pub draft: bool, + pub auto_merge: Option, + pub run_store: &'a RunStoreHandle, + pub llm_source: &'a dyn CredentialSource, + pub catalog: Arc, + pub conclusion: Option<&'a Conclusion>, + pub run_state: Option<&'a RunProjection>, } #[derive(Debug, Clone, PartialEq, Eq)] @@ -483,6 +468,7 @@ pub struct CreatedPullRequest { pub title: String, pub base_branch: String, pub head_branch: String, + pub head_sha: String, } /// Optionally open a pull request after a successful workflow run. @@ -516,6 +502,25 @@ pub async fn maybe_open_pull_request( let body = truncate_pr_body(&content.body); let title = content.title; + let remote_head = github_app::branch_head_sha(&req.github, &owner, &repo, req.head_branch) + .await + .map_err(|err| format!("failed to verify remote branch head: {err:#}"))?; + match remote_head { + Some(remote_head) if remote_head == req.expected_head_sha => {} + Some(remote_head) => { + return Err(format!( + "remote branch '{}' points to commit {remote_head}, expected final commit {}", + req.head_branch, req.expected_head_sha + )); + } + None => { + return Err(format!( + "remote branch '{}' does not exist; expected final commit {}", + req.head_branch, req.expected_head_sha + )); + } + } + let created = github_app::create_pull_request( &req.github, &owner, @@ -565,111 +570,10 @@ pub async fn maybe_open_pull_request( title, base_branch: req.base_branch.to_string(), head_branch: req.head_branch.to_string(), + head_sha: req.expected_head_sha.to_string(), })) } -/// PULL_REQUEST phase: optionally create a pull request after finalize. -/// -/// This stage is infallible: failures are emitted and logged, but the pipeline -/// completes. -pub async fn pull_request(concluded: Concluded, options: &PullRequestOptions) -> Finalized { - let Concluded { - outcome, - conclusion, - graph, - run_options, - services, - } = concluded; - - let mut pr_url = None; - if let Some(pr_cfg) = &options.pr_config { - if run_options.dry_run_enabled() { - tracing::debug!("Skipping PR creation: run is in dry-run mode"); - } else if let Err(ref e) = outcome { - tracing::debug!(error = %e, "Skipping PR creation: engine returned an error"); - } else if let Ok(ref result) = outcome { - if matches!( - result.status, - StageOutcome::Succeeded | StageOutcome::PartiallySucceeded - ) { - let diff = load_pull_request_diff(&services.run_store).await; - if let (Some(base_branch), Some(run_branch), Some(creds), Some(origin)) = ( - &run_options.base_branch, - run_options.run_branch(), - &options.github_app, - &options.origin_url, - ) { - let auto_merge = if pr_cfg.auto_merge { - Some(AutoMergeOptions { - merge_strategy: pr_cfg.merge_strategy, - }) - } else { - None - }; - - match maybe_open_pull_request(OpenPullRequestRequest { - github: github_app::GitHubContext::new( - creds, - &github_app::github_api_base_url(), - ), - origin_url: origin, - base_branch, - head_branch: run_branch, - goal: graph.goal(), - diff: &diff, - model: &options.model, - draft: pr_cfg.draft, - auto_merge, - run_store: &services.run_store, - llm_source: services.llm_source.as_ref(), - catalog: Arc::clone(&services.catalog), - conclusion: Some(&conclusion), - run_state: None, - }) - .await - { - Ok(Some(created)) => { - services.emitter.emit(&Event::pull_request_created( - &created.link, - &created.base_branch, - &created.head_branch, - &created.title, - pr_cfg.draft, - )); - pr_url = Some(created.link.html_url()); - } - Ok(None) => {} - Err(e) => { - services - .emitter - .emit(&Event::PullRequestFailed { error: e.clone() }); - services.emitter.notice( - RunNoticeLevel::Warn, - RunNoticeCode::PullRequestFailed, - format!("PR creation failed: {e}"), - ); - } - } - } - } - } - } - - Finalized { - run_id: run_options.run_id, - outcome, - conclusion, - pushed_branch: run_options - .settings - .run - .run_branch - .push - .then(|| run_options.run_branch().map(str::to_string)) - .flatten(), - pr_url, - } -} - #[cfg(test)] mod tests { use std::collections::HashMap; @@ -691,18 +595,14 @@ mod tests { }; use fabro_vault::{SecretType, Vault}; use futures::stream; - use httpmock::Method::POST; + use httpmock::Method::{GET, POST}; use httpmock::MockServer; use object_store::memory::InMemory; use tokio::sync::RwLock as AsyncRwLock; - use tokio_util::sync::CancellationToken; use super::*; use crate::event::{Event, append_event}; - use crate::outcome::Outcome; use crate::records::StageSummary; - use crate::run_options::{GitCheckpointOptions, RunOptions}; - use crate::services::EngineServices; struct MockProvider { name: String, @@ -917,48 +817,6 @@ mod tests { } } - #[tokio::test] - async fn pull_request_omits_pushed_branch_when_run_branch_push_disabled() { - let temp = tempfile::tempdir().unwrap(); - let mut settings = WorkflowSettings::default(); - settings.run.run_branch.push = false; - let run_options = RunOptions { - settings, - run_dir: temp.path().to_path_buf(), - cancel_token: CancellationToken::new(), - run_id: fixtures::RUN_1, - labels: HashMap::new(), - workflow_slug: None, - github_app: None, - pre_run_git: None, - fork_source_ref: None, - base_branch: None, - display_base_sha: None, - git: Some(GitCheckpointOptions { - base_sha: None, - run_branch: Some("fabro/run/test".to_string()), - meta_branch: None, - }), - }; - let concluded = Concluded { - outcome: Ok(Outcome::success()), - conclusion: make_test_conclusion(), - graph: Graph::new("test"), - run_options, - services: EngineServices::test_default().run, - }; - - let finalized = pull_request(concluded, &PullRequestOptions { - pr_config: None, - github_app: None, - origin_url: None, - model: "test-model".to_string(), - }) - .await; - - assert_eq!(finalized.pushed_branch, None); - } - // ── format_arc_details_section tests ──────────────────────────────── #[test] @@ -1555,20 +1413,21 @@ mod tests { }); let base_url = github_app::github_api_base_url(); let result = maybe_open_pull_request(OpenPullRequestRequest { - github: github_app::GitHubContext::new(&creds, &base_url), - origin_url: "https://github.com/owner/repo.git", - base_branch: "main", - head_branch: "fabro/run/123", - goal: "Fix bug", - diff: "", - model: "claude-sonnet-4-20250514", - draft: false, - auto_merge: None, - run_store: &run_store_handle, - llm_source: llm_source.as_ref(), - catalog: test_catalog(), - conclusion: None, - run_state: None, + github: github_app::GitHubContext::new(&creds, &base_url), + origin_url: "https://github.com/owner/repo.git", + base_branch: "main", + head_branch: "fabro/run/123", + expected_head_sha: "final-sha", + goal: "Fix bug", + diff: "", + model: "claude-sonnet-4-20250514", + draft: false, + auto_merge: None, + run_store: &run_store_handle, + llm_source: llm_source.as_ref(), + catalog: test_catalog(), + conclusion: None, + run_state: None, }) .await; assert!(result.is_ok()); @@ -1576,79 +1435,41 @@ mod tests { } #[tokio::test] - async fn load_pull_request_diff_uses_store_without_disk_patch() { - let tmp = tempfile::tempdir().unwrap(); - let store = test_store(); - let run_store = store.create_run(&fixtures::RUN_1).await.unwrap(); - let run_spec = RunSpec { - run_id: fixtures::RUN_1, - settings: fabro_types::WorkflowSettings::default(), - graph: Graph::new("test"), - graph_source: None, - workflow_slug: None, - automation: None, - source_directory: Some(tmp.path().display().to_string()), - git: None, - labels: std::collections::HashMap::new(), - provenance: test_support::test_run_provenance(), - manifest_blob: None, - definition_blob: None, - fork_source_ref: None, - }; - append_event(&run_store, &fixtures::RUN_1, &Event::RunCreated { - run_id: fixtures::RUN_1, - title: None, - settings: serde_json::to_value(&run_spec.settings).unwrap(), - graph: serde_json::to_value(&run_spec.graph).unwrap(), - workflow_source: None, - workflow_config: None, - labels: run_spec.labels.clone().into_iter().collect(), - run_dir: tmp.path().display().to_string(), - source_directory: run_spec.source_directory.clone(), - workflow_slug: None, - automation: None, - db_prefix: None, - provenance: run_spec.provenance.clone(), - manifest_blob: None, - git: None, - fork_source_ref: None, - retried_from: None, - parent_id: None, - web_url: None, + async fn stale_remote_branch_is_rejected_before_pull_request_creation() { + let payload = pr_content_json("Fix bug", "Narrative."); + let harness = setup_fallback_test_harness_with_branch_sha(&payload, "stale-sha").await; + let github_base_url = harness.github_server.url(""); + let error = maybe_open_pull_request(OpenPullRequestRequest { + github: fabro_github::GitHubContext::new(&harness.creds, &github_base_url), + origin_url: "https://github.com/owner/repo.git", + base_branch: "main", + head_branch: "fabro/run/123", + expected_head_sha: "final-sha", + goal: "Fix bug", + diff: "diff --git a/src/lib.rs b/src/lib.rs\n+fn x() {}\n", + model: "claude-sonnet-4-20250514", + draft: false, + auto_merge: None, + run_store: &harness.run_store, + llm_source: harness.llm_source.as_ref(), + catalog: harness.catalog.clone(), + conclusion: None, + run_state: None, }) .await - .unwrap(); - append_event(&run_store, &fixtures::RUN_1, &Event::RunRunnable { - source: fabro_types::RunRunnableSource::StartRequested, - actor: None, - }) - .await - .unwrap(); - append_event(&run_store, &fixtures::RUN_1, &Event::RunStarting) - .await - .unwrap(); - append_event(&run_store, &fixtures::RUN_1, &Event::RunRunning) - .await - .unwrap(); - append_event(&run_store, &fixtures::RUN_1, &Event::WorkflowRunCompleted { - timing: fabro_types::RunTiming::wall_only(1), - artifact_count: 0, - status: "succeeded".to_string(), - reason: SuccessReason::Completed, - total_usd_micros: None, - final_git_commit_sha: None, - final_patch: Some( - "diff --git a/src/lib.rs b/src/lib.rs\n+fn from_store() {}\n".to_string(), - ), - diff_summary: None, - billing: None, - }) - .await - .unwrap(); + .expect_err("stale remote branch must prevent PR creation"); - let diff = load_pull_request_diff(&run_store.clone().into()).await; - - assert!(diff.contains("from_store")); + assert!(error.contains("stale-sha")); + assert!(error.contains("final-sha")); + httpmock::Mock::new(harness.openai_mock_id, &harness.openai_server) + .assert_async() + .await; + httpmock::Mock::new(harness.branch_mock_id, &harness.github_server) + .assert_async() + .await; + httpmock::Mock::new(harness.github_mock_id, &harness.github_server) + .assert_calls_async(0) + .await; } // ── Structured-output PR content tests ────────────────────────────── @@ -1807,6 +1628,7 @@ mod tests { openai_server: MockServer, github_server: MockServer, openai_mock_id: usize, + branch_mock_id: usize, github_mock_id: usize, llm_source: Arc, catalog: Arc, @@ -1819,6 +1641,9 @@ mod tests { httpmock::Mock::new(self.openai_mock_id, &self.openai_server) .assert_async() .await; + httpmock::Mock::new(self.branch_mock_id, &self.github_server) + .assert_async() + .await; httpmock::Mock::new(self.github_mock_id, &self.github_server) .assert_async() .await; @@ -1830,6 +1655,13 @@ mod tests { /// credential source, and a run store seeded with a non-empty /// `final_patch`. async fn setup_fallback_test_harness(openai_payload_text: &str) -> FallbackHarness { + setup_fallback_test_harness_with_branch_sha(openai_payload_text, "final-sha").await + } + + async fn setup_fallback_test_harness_with_branch_sha( + openai_payload_text: &str, + branch_sha: &str, + ) -> FallbackHarness { let openai_server = MockServer::start_async().await; let openai_mock = openai_server .mock_async(|when, then| { @@ -1843,6 +1675,19 @@ mod tests { .await; let github_server = MockServer::start_async().await; + let branch_sha = branch_sha.to_string(); + let branch_mock = github_server + .mock_async(move |when, then| { + when.method(GET) + .path("/repos/owner/repo/branches/fabro/run/123") + .header("authorization", "Bearer test-token"); + then.status(200) + .header("content-type", "application/json") + .json_body(serde_json::json!({ + "commit": { "sha": branch_sha } + })); + }) + .await; let github_mock = github_server .mock_async(|when, then| { when.method(POST) @@ -1878,8 +1723,7 @@ mod tests { let store = test_store(); let run_store = store.create_run(&fixtures::RUN_1).await.unwrap(); - // Seed a non-empty `final_patch` so `load_pull_request_diff` returns - // diff content and the early-return for empty diffs does not fire. + // Seed a completed run so the PR body can include run details. let run_spec = RunSpec { run_id: fixtures::RUN_1, settings: fabro_types::WorkflowSettings::default(), @@ -1947,6 +1791,7 @@ mod tests { .unwrap(); let openai_mock_id = openai_mock.id; + let branch_mock_id = branch_mock.id; let github_mock_id = github_mock.id; FallbackHarness { @@ -1954,6 +1799,7 @@ mod tests { openai_server, github_server, openai_mock_id, + branch_mock_id, github_mock_id, llm_source, catalog, @@ -1978,6 +1824,7 @@ mod tests { origin_url: "https://github.com/owner/repo.git", base_branch: "main", head_branch: "fabro/run/123", + expected_head_sha: "final-sha", goal: "Fix telemetry leak\n\ndetails...", diff: "diff --git a/src/lib.rs b/src/lib.rs\n+fn x() {}\n", model: "gpt-5.4", @@ -2015,6 +1862,7 @@ mod tests { origin_url: "https://github.com/owner/repo.git", base_branch: "main", head_branch: "fabro/run/123", + expected_head_sha: "final-sha", goal: &goal, diff: "diff --git a/src/lib.rs b/src/lib.rs\n+fn x() {}\n", model: "gpt-5.4", diff --git a/lib/components/fabro-workflow/src/pipeline/types.rs b/lib/components/fabro-workflow/src/pipeline/types.rs index 425cba9aa..c7cfcc815 100644 --- a/lib/components/fabro-workflow/src/pipeline/types.rs +++ b/lib/components/fabro-workflow/src/pipeline/types.rs @@ -337,17 +337,59 @@ pub struct Executed { pub model: String, } -/// Output of the FINALIZE phase. +/// Output of the CONCLUDE phase. #[non_exhaustive] pub struct Concluded { - pub outcome: Result, - pub conclusion: Conclusion, - pub graph: Graph, - pub run_options: RunOptions, - pub services: Arc, + pub outcome: Result, + pub conclusion: Conclusion, + pub artifact_count: usize, + pub graph: Graph, + pub run_options: RunOptions, + pub services: Arc, } -/// Output of the PULL_REQUEST phase. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PublishOutcome { + NotRequested, + NoChanges { + pushed_branch: String, + }, + Published { + pushed_branch: String, + pr_url: Option, + }, +} + +impl PublishOutcome { + pub fn pushed_branch(&self) -> Option<&str> { + match self { + Self::NotRequested => None, + Self::NoChanges { pushed_branch } | Self::Published { pushed_branch, .. } => { + Some(pushed_branch) + } + } + } + + pub fn pr_url(&self) -> Option<&str> { + match self { + Self::Published { pr_url, .. } => pr_url.as_deref(), + Self::NotRequested | Self::NoChanges { .. } => None, + } + } +} + +/// Output of the PUBLISH phase. +#[non_exhaustive] +pub struct Published { + pub execution_outcome: Result, + pub publish_outcome: Result, + pub conclusion: Conclusion, + pub artifact_count: usize, + pub run_options: RunOptions, + pub services: Arc, +} + +/// Output of the FINALIZE phase. #[non_exhaustive] pub struct Finalized { pub run_id: RunId, @@ -383,8 +425,8 @@ pub struct FinalizeOptions { pub last_git_sha: Option, } -/// Options for the PULL_REQUEST phase. -pub struct PullRequestOptions { +/// Options for the PUBLISH phase. +pub struct PublishOptions { pub pr_config: Option, pub github_app: Option, pub origin_url: Option, diff --git a/lib/foundation/fabro-api/tests/status_round_trip.rs b/lib/foundation/fabro-api/tests/status_round_trip.rs index efaa5e7b0..7308b26fc 100644 --- a/lib/foundation/fabro-api/tests/status_round_trip.rs +++ b/lib/foundation/fabro-api/tests/status_round_trip.rs @@ -113,6 +113,7 @@ fn success_reason_json_tokens_match_openapi() { #[test] fn failure_reason_json_tokens_match_openapi() { assert_string_json(FailureReason::WorkflowError, "workflow_error"); + assert_string_json(FailureReason::PublishFailed, "publish_failed"); assert_string_json(FailureReason::Cancelled, "cancelled"); assert_string_json(FailureReason::ApprovalDenied, "approval_denied"); assert_string_json(FailureReason::Terminated, "terminated"); diff --git a/lib/foundation/fabro-types/src/run_event/misc.rs b/lib/foundation/fabro-types/src/run_event/misc.rs index d3bf9c1c0..41bf2cbbb 100644 --- a/lib/foundation/fabro-types/src/run_event/misc.rs +++ b/lib/foundation/fabro-types/src/run_event/misc.rs @@ -247,6 +247,8 @@ pub struct PullRequestCreatedProps { pub repo: String, pub base_branch: String, pub head_branch: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub head_sha: Option, pub title: String, pub draft: bool, } diff --git a/lib/foundation/fabro-types/src/status.rs b/lib/foundation/fabro-types/src/status.rs index 770f6aa5d..2224d0708 100644 --- a/lib/foundation/fabro-types/src/status.rs +++ b/lib/foundation/fabro-types/src/status.rs @@ -290,6 +290,7 @@ pub enum SuccessReason { #[strum(serialize_all = "snake_case")] pub enum FailureReason { WorkflowError, + PublishFailed, Cancelled, ApprovalDenied, Terminated, diff --git a/lib/packages/fabro-api-client/src/models/failure-reason.ts b/lib/packages/fabro-api-client/src/models/failure-reason.ts index b11a85248..79172e887 100644 --- a/lib/packages/fabro-api-client/src/models/failure-reason.ts +++ b/lib/packages/fabro-api-client/src/models/failure-reason.ts @@ -20,6 +20,7 @@ export const FailureReason = { WORKFLOW_ERROR: 'workflow_error', + PUBLISH_FAILED: 'publish_failed', CANCELLED: 'cancelled', APPROVAL_DENIED: 'approval_denied', TERMINATED: 'terminated', From 6efba896f4e52e46da51628ed71f0c07d4522636 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 11:58:54 -0400 Subject: [PATCH 12/76] Add Chisel quality calibration --- .chisel/calibration/calibration.md | 144 +++++ .chisel/calibration/work/adjudication.md | 249 +++++++++ .../calibration/work/consistency-review.md | 112 ++++ .chisel/calibration/work/reviewer-1.md | 182 +++++++ .chisel/calibration/work/reviewer-2.md | 176 +++++++ .chisel/calibration/work/reviewer-3.md | 490 ++++++++++++++++++ .chisel/calibration/work/validation-1.md | 424 +++++++++++++++ .chisel/calibration/work/validation-2.md | 450 ++++++++++++++++ .chisel/calibration/work/validation-3.md | 447 ++++++++++++++++ 9 files changed, 2674 insertions(+) create mode 100644 .chisel/calibration/calibration.md create mode 100644 .chisel/calibration/work/adjudication.md create mode 100644 .chisel/calibration/work/consistency-review.md create mode 100644 .chisel/calibration/work/reviewer-1.md create mode 100644 .chisel/calibration/work/reviewer-2.md create mode 100644 .chisel/calibration/work/reviewer-3.md create mode 100644 .chisel/calibration/work/validation-1.md create mode 100644 .chisel/calibration/work/validation-2.md create mode 100644 .chisel/calibration/work/validation-3.md diff --git a/.chisel/calibration/calibration.md b/.chisel/calibration/calibration.md new file mode 100644 index 000000000..a5777e0c6 --- /dev/null +++ b/.chisel/calibration/calibration.md @@ -0,0 +1,144 @@ +# Chisel Quality Calibration + +Calibration v1 · cartography v1 · revision `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` · 2026-07-27T15:55:07Z +Sample: `fabro-workflow`, `fabro-http`, `fabro-web-app`, `repository-ci` · Control: `fabro-checkpoint` at `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` +Evaluators: GPT-5 (Codex primary and independent reviewers) + +## How to Use This Calibration + +Judge each mapped component against its purpose and direct repository evidence. +Do not grade on a curve. Apply one score per lens, and count a finding under +only its primary lens. + +**Isolated** means contained at an edge; normal callers and routine changes do +not encounter it. **Central** means part of a mapped entry point, common path, +or recurring change. A **routine change** is an ordinary extension or +maintenance task implied by the component's mapped purpose. + +Infer routine work from the mapped purpose and traced common paths; a public +method alone does not establish frequency. A directly evidenced central concern +caps the component's lens score rather than being averaged against healthier +sub-responsibilities. Necessary delegation inside a clear owner is not pressure, +and size or internal busyness alone does not lower ownership. + +Use **N/E** when evidence is insufficient. Never convert missing evidence into +a numeric score, and do not penalize a missing lifecycle path without evidence +that the mapped purpose requires it. Score 4 requires a positive production +mechanism and no material friction; tests may corroborate that mechanism but +cannot create it or become a second authority merely by asserting its contract. + +## Lenses + +### `ownership-boundaries` — Ownership and boundaries + +**Does each responsibility and lifecycle have a clear home, with dependencies +pointing in the intended direction?** Includes responsibility, state, resource, +dependency, and lifecycle placement; excludes local control flow, naming, +types, API meaning, and repeated policy alone. + +### `simplicity` — Simplicity + +**Is the implementation no more complex, indirect, or general than necessary?** +Includes common-path traceability, control flow, indirection, abstraction, and +configuration burden; excludes placement, domain meaning, and independently +repeated knowledge. + +### `domain-model` — Domain model + +**Does each domain concept have one clear meaning and valid shape?** Includes +types, terminology, legal states, conversions, validation, and API semantics; +excludes module placement, lifecycle ownership, and repetition preserving one +meaning. + +### `duplication-knowledge` — Duplication of knowledge + +**Are policies, invariants, decisions, and transformations authoritative rather +than repeated?** Includes semantic repetition and manual synchronization; +excludes harmless syntax, coincidental similarity, and unification that would +create a parameterized mega-abstraction. + +## Observable Anchors + +| Score | Ownership and boundaries | Simplicity | Domain model | Duplication of knowledge | +|---:|---|---|---|---| +| 4 | One owner contains the mapped responsibility's state and complete lifecycle. | A production mechanism makes the necessary common path directly traceable. | Canonical types reject invalid states before every common-path interpretation. | One authoritative mechanism enforces each recurring policy, invariant, or transformation. | +| 3 | Ownership friction is isolated outside routine changes. | Unnecessary indirection is isolated outside routine changes. | Meaning or validation friction is isolated outside routine changes. | Repeated knowledge is isolated outside routine changes. | +| 2 | Routine changes coordinate competing owners or reverse the mapped dependency direction. | Routine changes repeatedly navigate competing paths, avoidable layers, or configuration machinery. | Routine changes reconcile recurring meanings, conversions, or invalid intermediate states. | Routine changes manually synchronize the same policy, invariant, or transformation across recurring locations. | +| 1 | No stable owner or dependency direction can be identified for the responsibility. | No stable common path can be traced through the implementation. | No stable meaning or legal shape can be identified for a core concept. | No stable authority can be identified for recurring domain knowledge. | + +## Decision Rules + +1. A directly evidenced central concern caps the component's lens score; do not average it against healthier sub-responsibilities. +2. Judge ownership against the map, not type names; when routine callers reconstruct a mapped lifecycle from low-level primitives, ownership fits 2. +3. A check owns trigger coverage for every path it scans; non-triggering routine targets are ownership pressure, while nonexistent selector values are domain-model pressure. +4. An unused production dependency or parallel entry layer is isolated simplicity friction, capping 4 at 3 when the common path remains direct. +5. Caller validation or a typed destination does not isolate an invalid-capable mapped entry; routine common-path use of that shape fits 2. +6. Concrete second semantic representations cap 4 at 3; score 2 only when an ordinary mapped change must synchronize them, not merely because call sites repeat. + +## Confidence + +Confidence describes evidence quality, not severity. **High** requires direct +evidence across relevant common and boundary paths; final High also requires +independent readings to converge. **Medium** has a material ambiguity or +coverage gap. **Low** is partial or substantially inferential. + +## Classifying a Finding + +- Where should this responsibility or lifecycle live? → `ownership-boundaries` +- Why is this much machinery necessary? → `simplicity` +- What does this name, type, state, or API value mean? → `domain-model` +- Why is this knowledge authoritative in several places? → `duplication-knowledge` + +Tags are diagnostic metadata, not additional scores: + +```text +abstraction-burden boundary-leakage configuration-sprawl +control-flow conversion-sprawl dependency-direction +generality indirection invalid-states +lifecycle misplaced-responsibility +ownership repeated-invariant repeated-policy +repeated-test-knowledge repeated-transformation +state-coupling type-sprawl vocabulary-drift +``` + +## Repository Examples + +### `ownership-boundaries` + +- `lib/components/fabro-workflow/src/lifecycle/mod.rs:WorkflowLifecycle` shows a central orchestrator can own callback order through focused delegates; reviewers must still inspect terminal paths before calling lifecycle ownership contained. +- `apps/fabro-web/app/lib/api-client.ts:apiData` and `apps/fabro-web/app/lib/queries.ts:useRun` keep shared transport and read lifecycles out of route composition; a busy route alone is not boundary leakage. + +### `simplicity` + +- `lib/foundation/fabro-http/src/lib.rs:define_builder!` makes async and blocking construction traceable through one necessary mechanism; local macro indirection can reinforce simplicity. +- `lib/components/fabro-workflow/src/operations/start.rs:RunSession::run` exposes a linear phase sequence, while service reshaping across phase inputs shows that a stable path can still carry recurring machinery. + +### `domain-model` + +- `lib/components/fabro-workflow/src/event/events.rs:Event::StageCompleted` uses string status before `lib/components/fabro-workflow/src/event/convert.rs:stage_status_from_string` reparses it; a typed durable result does not isolate this common-path intermediate. +- `lib/foundation/fabro-http/src/lib.rs:ProxyPolicy` and `ProxyPolicy::resolve_with_env_value` demonstrate a closed policy vocabulary whose invalid boundary values are rejected. + +### `duplication-knowledge` + +- `lib/components/fabro-workflow/src/event/names.rs:event_name` and `lib/components/fabro-workflow/src/event/convert.rs:event_body_from_event` show manual mappings that a routine event extension must synchronize, even when exhaustive matches detect omissions. +- `.github/workflows/rust.yml:on.push.paths` and `.github/workflows/rust.yml:on.pull_request.paths` demonstrate duplicated trigger knowledge: one source-area change requires two manual policy edits. + +## Control Baseline + +`fabro-checkpoint` at `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c`: + +| Lens | Score | Confidence | +|---|---:|---| +| Ownership and boundaries | 2 | High | +| Simplicity | 3 | High | +| Domain model | 2 | Medium | +| Duplication of knowledge | 3 | Medium | + +## Recalibration Triggers + +Recalibrate only for a rubric change, a material cartography change, a model +change with demonstrated drift, or inconsistent scores on the control sample. + +## Open Questions + +None. diff --git a/.chisel/calibration/work/adjudication.md b/.chisel/calibration/work/adjudication.md new file mode 100644 index 000000000..8f3f7bb28 --- /dev/null +++ b/.chisel/calibration/work/adjudication.md @@ -0,0 +1,249 @@ +# Calibration Adjudication + +Revision: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +Cartography: v1 at `2bcf94fed8a9b429f18d9196fa824711d6f4cb0a`. +The only later commit adds cartography artifacts, so the mapped code paths are +unchanged at the assessed revision. + +Sample: `fabro-workflow`, `fabro-http`, `fabro-web-app`, `repository-ci`. +Control: `fabro-checkpoint`. + +## Independent Score Matrix + +Cells list reviewer 1 / reviewer 2 / reviewer 3. + +| Component | Ownership and boundaries | Simplicity | Domain model | Duplication of knowledge | +|---|---:|---:|---:|---:| +| `fabro-workflow` | 2 / 3 / 4 | 2 / 2 / 2 | 2 / 3 / 2 | 2 / 2 / 2 | +| `fabro-http` | 4 / 4 / 4 | 4 / 4 / 4 | 4 / 3 / 4 | 3 / 4 / 4 | +| `fabro-web-app` | 4 / 4 / 3 | 2 / 2 / 2 | 2 / 2 / 2 | 2 / 2 / 2 | +| `repository-ci` | 4 / 4 / 3 | 3 / 3 / 3 | 2 / 3 / 2 | 2 / 2 / 2 | + +Unanimous pairs establish that central machinery may still have a stable path: +`fabro-workflow` is 2 for simplicity and duplication; `fabro-web-app` is 2 for +simplicity, domain model, and duplication; and `repository-ci` is 3 for +simplicity and 2 for duplication. `fabro-http` is unanimously 4 for ownership +and simplicity. + +## Material Disagreements + +### `fabro-workflow` × ownership and boundaries — 2 / 3 / 4 + +- **Evidence:** `pipeline/mod.rs` and `pipeline/types.rs` give the normal run + explicit phase owners; `lifecycle/mod.rs:WorkflowLifecycle` owns callback + ordering through focused delegates. +- **Counterevidence:** terminal completion and failure are also constructed in + `pipeline/finalize.rs:build_terminal_event`, + `operations/start.rs:emit_workflow_run_failed`, + `operations/start.rs:persist_terminal_engine_failure`, completion/drop + guards, retry, and archive operations. +- **Ambiguous rule:** two reviewers judged the clear normal path; one judged + whether the same lifecycle has one home across normal and exceptional paths. +- **Discriminator:** inspect every recurring terminal path. A routine + terminal-contract change crossing several operation owners is score-2 + ownership pressure even when the success path is well partitioned. +- **Draft adjudication:** 2. + +### `fabro-workflow` × domain model — 2 / 3 / 2 + +- **Evidence:** `pipeline/types.rs` encodes phase states and canonical product + records are reused. +- **Counterevidence:** `event/events.rs:Event::StageCompleted` carries a string + status; `lifecycle/event.rs:EventLifecycle::after_node` serializes a typed + outcome and `event/convert.rs:stage_status_from_string` reparses it with an + unknown-value fallback. +- **Ambiguous rule:** whether a typed durable event isolates an invalid + intermediate representation on the common producer path. +- **Discriminator:** common-path invalid intermediate states are central even + when the durable result is typed. +- **Draft adjudication:** 2. + +### `fabro-http` × domain model — 4 / 3 / 4 + +- **Evidence:** `ProxyPolicy`, `resolve_with_env_value`, and + `HttpClientBuildError` form a closed policy with explicit precedence and + rejection. +- **Counterevidence:** public builders expose both + `proxy_policy(ProxyPolicy::Disabled)` and lower-level `no_proxy()`. +- **Ambiguous rule:** whether a lower-level transport control creates a second + meaning for the repository policy. +- **Discriminator:** an escape hatch does not split the canonical concept when + the typed policy remains closed and its precedence is enforced. +- **Draft adjudication:** 4. + +### `fabro-http` × duplication of knowledge — 3 / 4 / 4 + +- **Evidence:** `define_builder!` is the shared async/blocking authority and + `ProxyPolicy::resolve` owns precedence. +- **Counterevidence:** adding a policy variant synchronizes the enum, parser, + expected-value error text, behavior match, and tests. +- **Ambiguous rule:** whether co-location and exhaustive matching make all + policy vocabulary authoritative. +- **Discriminator:** hypothetical variants do not establish routine + recurrence; exhaustive compiler-checked behavior remains one authority + unless direct evidence shows recurring manual synchronization. +- **Draft adjudication:** 4. + +### `fabro-web-app` × ownership and boundaries — 4 / 4 / 3 + +- **Evidence:** `entry.tsx`, route graphs, `lib/api-client.ts`, queries, + mutations, effect hooks, and the build script give shared responsibilities + visible homes. +- **Counterevidence:** `install-app.tsx` and `routes/run-stages.tsx` contain + several central transformations and presentation concerns. +- **Ambiguous rule:** whether a busy but clearly identified route owner is + boundary pressure or simplicity pressure. +- **Discriminator:** do not lower ownership for internal complexity unless + routine changes cross another owner or reverse the mapped dependency + direction. +- **Draft adjudication:** 4. + +### `repository-ci` × ownership and boundaries — 4 / 4 / 3 + +- **Evidence:** Rust and TypeScript workflows have distinct validation jobs, + narrow permissions, and delegate build procedures to repository commands. +- **Counterevidence:** the Rust clippy job embeds the repository's legacy-auth + vocabulary check. +- **Ambiguous rule:** whether enforcement of a product migration invariant is + misplaced when CI owns validation but not the underlying vocabulary. +- **Discriminator:** a named invariant check may live in CI, but its product + vocabulary must remain authoritative elsewhere; this isolated boundary + friction fits 3. +- **Draft adjudication:** 3. + +### `repository-ci` × domain model — 2 / 3 / 2 + +- **Evidence:** job, runner, permission, and test-mode vocabulary is otherwise + coherent. +- **Counterevidence:** `rust.yml:on.*.paths` names nonexistent `openapi/**` + rather than `docs/public/api-reference/fabro-api.yaml`, and + `zizmor.yml:rules.stale-action-refs.ignore` identifies exceptions by stale + line positions. +- **Ambiguous rule:** whether configuration references are domain vocabulary + or only duplicated operational data. +- **Discriminator:** identifiers that control central behavior are domain + vocabulary; missing or stale referents create score-2 pressure. +- **Draft adjudication:** 2. + +## Draft Anchor Decisions + +- Anchor score 4 on a positive enforcing mechanism, never absence of a defect. +- Separate owner clarity from the amount of machinery inside that owner. +- Treat invalid common-path intermediate states as domain-model pressure. +- Treat repeated semantic decisions as duplication only when routine changes + require manual synchronization. +- Treat mapped configuration identifiers as domain vocabulary. +- Reserve N/E for a lens without direct evidence; no sampled pair required it. + +## Consistency Review + +The fresh reviewer applied only the written draft to `fabro-checkpoint` and +reported: + +| Lens | Score | Evidence confidence | +|---|---:|---| +| Ownership and boundaries | 2 | Medium | +| Simplicity | 3 | High | +| Domain model | 2 | High | +| Duplication of knowledge | 2 | High | + +The control exposed four material wording problems: + +1. The draft did not say how a component-level score combines several + responsibilities, or whether positive mechanisms and friction can coexist + at score 4. +2. Necessary layered delegation could satisfy the original ownership and + simplicity score-2 wording. +3. The domain rules did not say when a public low-level API is an escape hatch + or what score a common invalid intermediate implies. +4. Decision rule 5 contradicted the duplication anchor by assigning routine + string synchronization to score 3. + +The revision now says that a central concern caps rather than averages, score 4 +requires a positive production mechanism without material friction, public +surface alone does not establish routine work, and missing paths are not +negative without mapped-purpose evidence. The anchors now distinguish competing +owners from necessary delegation and maintainer navigation from runtime +layering. Decision rules 2–6 resolve scoped lifecycle handoff, necessary +delegation, common-path invalid states, direct evidence of recurring +synchronization, and configuration identifiers. Tests corroborate production +authorities but are not second authorities merely because they restate a +contract. + +All 16 wording observations in `consistency-review.md` are covered by those +changes or by the existing primary-lens and confidence sections. No consistency +objection remains open before validation. + +## Validation + +### Round 1 + +| Assignment | Validator 1 | Validator 2 | Validator 3 | Result | +|---|---:|---:|---:|---| +| `fabro-workflow` × ownership | 2 | 2 | 2 | Resolved | +| `fabro-workflow` × domain | 2 | 2 | 2 | Resolved | +| `fabro-http` × domain | 4 | 4 | 4 | Resolved | +| `fabro-http` × duplication | 4 | 4 | 3 | Repeated adjacent split | +| `fabro-web-app` × ownership | 4 | 4 | 4 | Resolved | +| `repository-ci` × ownership | 4 | 2 | 4 | Non-adjacent split | +| `repository-ci` × domain | 2 | 2 | 2 | Resolved | +| Control × ownership | 2 | 4 | 2 | Non-adjacent split | +| Control × simplicity | 4 | 4 | 3 | Adjacent split | +| Control × domain | 2 | 3 | 2 | Adjacent split | +| Control × duplication | 3 | 2 | 3 | Adjacent split | + +The sample's workflow lifecycle, event status, HTTP policy model, web +composition, and CI identifier anchors now converge. Six assignments require +the permitted final simplification: + +- HTTP diagnostic allowed-value text is a concrete second semantic + representation, even though the macro is the behavioral authority. +- A CI check owns trigger coverage for every path its embedded policy scans; + this is distinct from the domain meaning of a nonexistent selector. +- Control ownership is judged against the mapped metadata-branch purpose, not + against narrower names on `Store` and `BranchStore`. +- The control's unused dependency and unused parallel entry layer are isolated + simplicity friction rather than evidence-free public breadth. +- Validation in an external caller does not make an invalid-capable mapped + entry type enforce its own legal shape. +- Repeated fixed Git protocol syntax is a concrete second representation, but + multiple current call sites alone do not make changing that protocol an + ordinary mapped change. + +Decision rules 2–6 now state those discriminators directly. Round 2 will +re-score only the six unresolved assignments. + +### Round 2 + +| Assignment | Validator 1 | Validator 2 | Validator 3 | Result | +|---|---:|---:|---:|---| +| `fabro-http` × duplication | 3 | 3 | 3 | Resolved | +| `repository-ci` × ownership | 2 | 2 | 2 | Resolved | +| Control × ownership | 2 | 2 | 2 | Resolved | +| Control × simplicity | 3 | 3 | 3 | Resolved | +| Control × domain | 2 | 2 | 2 | Resolved | +| Control × duplication | 3 | 3 | 3 | Resolved | + +All round-2 scores converge. The final control baseline is ownership 2 +(High), simplicity 3 (High), domain model 2 (Medium), and duplication of +knowledge 3 (Medium). Domain confidence remains Medium because one validator +found a material ambiguity over whether low-level Git path validation belongs +inside the component. Duplication confidence remains Medium because stable +protocol syntax is concrete repetition but has limited demonstrated change +burden. + +Across both validation rounds, the final disputed sample scores are: + +| Component | Ownership and boundaries | Domain model | Duplication of knowledge | +|---|---:|---:|---:| +| `fabro-workflow` | 2 | 2 | — | +| `fabro-http` | — | 4 | 3 | +| `fabro-web-app` | 4 | — | — | +| `repository-ci` | 2 | 2 | — | + +No non-adjacent or repeated adjacent split remains. + +## Open Questions + +None. diff --git a/.chisel/calibration/work/consistency-review.md b/.chisel/calibration/work/consistency-review.md new file mode 100644 index 000000000..e41ff2c26 --- /dev/null +++ b/.chisel/calibration/work/consistency-review.md @@ -0,0 +1,112 @@ +# Chisel Consistency Review: `fabro-checkpoint` + +Revision: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +Scope: `lib/components/fabro-checkpoint/**` only. The scored evidence is the manifest, production source, and unit tests at the pinned revision. I did not inspect callers, sample reviews, adjudication, or any other file under `.chisel/calibration/work/`. + +## Scores + +| Lens | Score | Confidence | +|---|---:|---| +| `ownership-boundaries` | 2 | Medium | +| `simplicity` | 3 | High | +| `domain-model` | 2 | High | +| `duplication-knowledge` | 2 | High | + +Confidence here describes this reading's evidence quality. The rubric's additional requirement that a final High confidence needs independent convergence can only be decided during adjudication. + +## `ownership-boundaries`: 2 + +The central branch lifecycle crosses two public owners. `BranchStore` stores the branch name and owns bootstrap plus normal branch reads and writes (`branch.rs:20-209`), but branch cleanup is exposed only as `Store::delete_ref(branch)` (`git.rs:215-226`). `BranchStore` keeps both its `Store` reference and branch name private and has no cleanup/archive operation. A caller therefore has to retain the same raw branch identity and leave the branch-scoped interface for cleanup. Bootstrap sequencing is also caller-owned: `BranchStore::new` does not establish the branch, writes fail when it is absent, and every writable test explicitly calls `ensure_branch` first (`branch.rs:26-81, 282-343`). This is recurring lifecycle work rather than an isolated edge, especially under decision rule 1. Primary tags: `lifecycle`, `ownership`. + +Strongest counterevidence: once initialized, `BranchStore::write_with` keeps the read-modify-write sequence together and delegates only Git object/ref primitives to `Store` (`branch.rs:56-82`). The dependency direction is stable: branch storage depends on the lower-level Git store, not vice versa. + +Why adjacent scores do not fit: + +- **1 does not fit:** `BranchStore` is a stable, identifiable owner for the common branch-scoped read/write responsibility, and `Store` is a coherent lower-level Git owner. +- **3 does not fit:** the split includes explicit bootstrap and cleanup paths. Decision rule 1 says recurring terminal ownership cannot be treated as isolated merely because the success path is clear. + +Confidence is Medium because the split is direct, but the scoped evidence cannot show whether archive, retry, and cleanup are deliberately owned by a higher-level caller. + +## `simplicity`: 3 + +The common write path is directly traceable: `write_entry`/`write_entries` prepare blobs, `write_with` reads the tip tree, applies one mutation, writes one commit, and advances one ref (`branch.rs:56-109`). `Store::read_tree` and `Store::write_tree` use a single flat `TreeEntries` representation with private recursive helpers (`git.rs:39-99, 141-159, 229-310`). These are positive reinforcing mechanisms, not just an absence of complexity. + +The remaining simplicity pressure is isolated configuration burden. The manifest declares `fabro-store`, `serde`, and the dev dependency `chrono` (`Cargo.toml:16-28`), but none is referenced anywhere in the component source or tests at this revision. The public `Store::repo` escape hatch (`git.rs:112-114`) and the lower-level object API also add surface area, but normal branch writes do not have to choose among competing implementations. Primary tag: `configuration-sprawl`. + +Strongest counterevidence to lowering the score: the component has one linear common mutation path, and its indirection corresponds directly to Git's blob/tree/commit/ref structure. + +Why adjacent scores do not fit: + +- **2 does not fit:** ordinary reads and writes do not repeatedly traverse competing orchestration paths or configuration machinery; the `BranchStore` to `Store` layering is stable and direct. +- **4 does not fit:** the centralized mutation path is a qualifying positive mechanism, but the unused manifest dependencies are concrete unnecessary configuration rather than necessary machinery. + +Confidence is High because all component files are in scope, so the dependency non-use and the full common write path are directly observable. + +## `domain-model`: 2 + +The common tree-entry producer accepts invalid intermediate path states. `TreeEntries` hides its map, but its public `set` accepts any `Into` without validating a relative Git path (`git.rs:46-61`). Both `BranchStore::write_entry` and `write_entries` feed caller-provided `&str` paths directly into it (`branch.rs:84-109`), and `build_dir_node` later assigns meaning by splitting the strings on `/` (`git.rs:270-294`). Empty components, leading/trailing separators, and file/directory prefix collisions are therefore representable in the canonical intermediate type and reach late Git-tree construction rather than being rejected at the common boundary. Branch identity is likewise an arbitrary `String` until `git2` receives the synthesized ref name (`branch.rs:20-38`, `git.rs:182-197`). This is central invalid-state pressure under decision rule 3, not an isolated low-level escape hatch. Primary tag: `invalid-states`. + +The small helper `sharded_path` is corroborating boundary evidence: its contract says the input is a hex ID, but its public signature accepts any `&str` and slices at a caller-provided byte offset (`branch.rs:211-220`), so a non-ASCII input can panic rather than be rejected as invalid input. + +Strongest counterevidence: `FileMode` is a closed enum and `TreeEntries` keeps ordering and representation private (`git.rs:13-99`). `Error` also distinguishes a missing branch from generic Git failures (`error.rs:5-18`). The component therefore has stable concepts even though common constructors do not preserve all their invariants. + +Why adjacent scores do not fit: + +- **1 does not fit:** branch storage, tree entries, file modes, authors, and trailers all have recognizable, stable meanings. +- **3 does not fit:** raw paths and branch names enter the common public read/write boundary, so validation friction is not isolated outside routine use. + +Confidence is High because the accepting producers and their downstream interpretation are both visible within the scoped common path. + +## `duplication-knowledge`: 2 + +The transformation “find a path in a commit tree, treat only `NotFound` as absence, load the entry as a blob, and copy its bytes” is independently implemented by `BranchStore::read_entry`, `BranchStore::read_entries`, and `Store::read_blob_at` (`branch.rs:119-158`, `git.rs:200-213`). An ordinary maintenance change to missing-entry or entry-kind behavior must synchronize all three common read locations. Ref qualification is also repeated in `update_ref`, `resolve_ref`, and `delete_ref` (`git.rs:182-226`). + +Trailer grammar supplies independent corroboration at the commit-message edge: `": "` formatting/detection is separately encoded by `append`, `parse`, `format_message`, and `has_trailing_trailer_block` (`trailer.rs:9-25, 28-42, 45-65, 68-87`). Primary tags: `repeated-transformation`, `repeated-policy`. + +Strongest counterevidence: important write knowledge is authoritative. `BranchStore::write_with` centralizes tip loading, parent linkage, commit creation, and ref advancement, while `GitAuthor::default` centralizes the fallback identity (`branch.rs:56-82`, `author.rs:13-35`). + +Why adjacent scores do not fit: + +- **1 does not fit:** the repeated implementations currently agree, and stable authorities exist for branch mutation, author defaults, and file-mode conversion. +- **3 does not fit:** the repeated blob-read transformation appears on the public latest-entry and multi-entry common paths, so a routine storage-policy change encounters it centrally rather than only at an edge. + +Confidence is High because the repeated transformations and the mechanisms that are already centralized can both be enumerated completely inside the scoped component. + +## Rubric wording audit + +The following rules or anchors were ambiguous or non-discriminating in this application. I resolved each explicitly rather than silently choosing an interpretation. + +1. **One component score across several responsibilities.** The instruction says to judge “each mapped component,” while the anchors use singular phrases such as “a mapped responsibility” and “a core concept.” It does not say whether to average sub-responsibilities, take the worst concern, or weight by centrality. I scored the mapped checkpoint-storage responsibility and let a directly evidenced central concern cap the lens; isolated author/trailer helpers could affect a score only at 3 versus 4. + +2. **How to establish “routine” and “central” with component-only evidence.** A public method may be a mapped entry point without being frequent, and scoped evidence cannot establish caller frequency. I treated bootstrap, latest reads/writes, and cleanup as routine because they are ordinary lifecycle operations implied by branch storage. I did not infer frequency for unrelated external call sites. + +3. **N/E threshold versus an absent lifecycle path.** “Use N/E when evidence is insufficient” does not say whether a missing archive/retry API is negative evidence, out of scope, or grounds for N/E. I scored paths that are directly present (bootstrap, normal operation, cleanup), did not penalize an unobserved archive/retry design, and lowered ownership confidence for the coverage gap. + +4. **Score 3 and score 4 overlap in every lens.** A positive reinforcing mechanism can coexist with isolated friction, so the score-4 requirement and score-3 anchor can both be true. I treated any evidenced unnecessary/frictional mechanism as a cap at 3; score 4 requires both a positive mechanism and no material friction in the mapped responsibility. This is why the unused manifest dependencies keep simplicity at 3 despite `write_with`. + +5. **What qualifies as a “positive reinforcing mechanism.”** The rubric does not say whether tests, encapsulation alone, or a production authority qualifies. I required an operative production mechanism that funnels behavior or rejects invalid construction. Tests alone did not qualify. + +6. **Ownership score 2 versus ordinary delegation.** “Cross recurring owners or dependency boundaries” could penalize every layered implementation. Decision rule 2 partly resolves this, but “same responsibility” remains subjective. I treated `BranchStore` calling `Store` during a write as ordinary delegation; I counted cleanup only because the caller must leave the branch-scoped owner and supply its identity again. + +7. **Decision rule 1 when terminal operations live at a lower abstraction.** The rule says not to isolate recurring terminal owners but does not define whether a lower-level deletion primitive is a second owner or a delegate. Because `BranchStore` offers no cleanup interface and keeps the needed state private, I treated `Store::delete_ref` as a lifecycle-owner crossing, not merely internal machinery. + +8. **Simplicity score 2’s “repeatedly traverse.”** It is unclear whether this means runtime calls passing through multiple necessary layers, or maintainers choosing among competing paths repeatedly. I used the latter interpretation, consistent with the lens question and decision rule 2; necessary Git layers did not lower the score. + +9. **Decision rule 2’s “simplicity pressure.”** The rule labels machinery inside an owner as pressure even though the lens expressly permits necessary complexity and gives no score consequence for “pressure.” I treated machinery as evidence to test for necessity, not as an automatic deduction. + +10. **Domain score 4 versus decision rule 4’s escape hatch.** “Every common boundary” is not defined, and a public low-level API can be called common or an escape hatch depending on external usage. I treated `TreeEntries::set` as common because `BranchStore::write_with`, `write_entry`, and `write_entries` use it directly; `Store::repo` was treated as an escape hatch. + +11. **Decision rule 3 does not identify a score boundary.** It says a typed durable value does not “repair domain pressure,” but does not say whether a common invalid intermediate means 2 or merely prevents 4. I mapped common-path invalid intermediates to the score-2 anchor (“routine changes reconcile ... invalid intermediate states”); isolated invalid intermediates would map to 3. + +12. **Duplication score 2 versus decision rule 5.** Rule 5 says to score 3 when a routine vocabulary change requires synchronization, while the score-2 anchor says routine synchronization of the same policy/invariant/transformation is score 2. Those statements conflict unless “vocabulary” is an unstated special case. I treated rule 5 narrowly as an exception for localized, string-only vocabulary at an edge. The score-2 finding here rests instead on repeated behavioral blob-read transformations on common paths. + +13. **What test repetition counts as knowledge duplication.** The `repeated-test-knowledge` tag suggests tests can count, but the anchors do not distinguish duplicated policy from assertions that intentionally restate expected behavior. I did not count an assertion of a production contract as a second authority. Repeated test fixture setup was only isolated counterevidence and did not drive a numeric score. + +14. **Decision rule 6 lacks a lens and defines neither “current referent” nor “line selector.”** Its opening phrase points toward `domain-model`, while duplicated CI selectors could point toward `duplication-knowledge`; its mandatory score 2 also bypasses centrality analysis. It had no referent in this component, so I did not apply it. If applicable, I would classify a single invalid identifier under domain model and synchronized copies under duplication. + +15. **The “primary lens only” rule does not explain multi-causal facts.** Raw strings can simultaneously expose invalid states, repeat vocabulary, and force lifecycle handoffs. I assigned each negative fact once by its primary question: lifecycle handoff to ownership, unused dependencies to simplicity, raw path legality to domain, and repeated lookup/ref/trailer behavior to duplication. + +16. **Confidence High cannot be finalized by one reviewer.** “Final High also requires independent readings to converge” is not decidable during an independent review. I reported evidence-quality confidence now and left final convergence to adjudication. + +All other score-1 versus score-2 distinctions were discriminating here: the component consistently has identifiable owners, paths, concepts, and intended policies, so none of the “no stable ... can be identified” anchors fit. diff --git a/.chisel/calibration/work/reviewer-1.md b/.chisel/calibration/work/reviewer-1.md new file mode 100644 index 000000000..d6c0140da --- /dev/null +++ b/.chisel/calibration/work/reviewer-1.md @@ -0,0 +1,182 @@ +# Calibration review — reviewer 1 + +Revision reviewed: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +Scope: `fabro-workflow`, `fabro-http`, `fabro-web-app`, and `repository-ci` as routed by `.chisel/cartography/codebase-map.md`. I excluded `apps/fabro-web/app/components/playground/**` from `fabro-web-app`, and limited `repository-ci` to `.github/workflows/rust.yml`, `.github/workflows/typescript.yml`, and `.github/zizmor.yml`. The routed paths have no changes between the map revision and the reviewed revision. + +## Provisional ratings + +| Component | Ownership boundaries | Simplicity | Domain model | Duplication of knowledge | +| --- | --- | --- | --- | --- | +| `fabro-workflow` | **2 — High** | **2 — High** | **2 — High** | **2 — High** | +| `fabro-http` | **4 — High** | **4 — High** | **4 — High** | **3 — High** | +| `fabro-web-app` | **4 — High** | **2 — High** | **2 — High** | **2 — High** | +| `repository-ci` | **4 — High** | **3 — High** | **2 — High** | **2 — High** | + +## `fabro-workflow` + +### Ownership boundaries — 2, High confidence + +The component has a clear top-level phase boundary: `pipeline/mod.rs` orders parse, transform, validate, initialize, execute, finalize, and pull-request processing; `pipeline/types.rs` gives those phases distinct result types. `pipeline/execute.rs:execute`, `graph.rs:WorkflowGraph`, and `node_handler.rs:WorkflowNodeHandler` also make the boundary with the generic `fabro-core` executor explicit. `lifecycle/mod.rs:WorkflowLifecycle` composes named lifecycle owners instead of placing every callback in the executor. + +The pressure appears in terminal-run ownership. The normal path is owned by `pipeline/finalize.rs:finalize` and `pipeline/finalize.rs:build_terminal_event`, while engine/bootstrap failures are handled by `operations/start.rs:emit_workflow_run_failed`, `operations/start.rs:persist_terminal_engine_failure`, and the completion/drop guards in `operations/start.rs`. Retry and archive operations also synthesize terminal events in `operations/retry.rs` and `operations/archive.rs`. These paths are understandable individually, but terminal state, persistence, and event emission do not have one stable lifecycle home. + +A representative routine change is adding terminal metadata that must be present for every failed or concluded run. It would require checking or changing `pipeline/finalize.rs:build_terminal_event`, `pipeline/finalize.rs:finalize`, `operations/start.rs:emit_workflow_run_failed`, `operations/start.rs:persist_terminal_engine_failure`, the start-operation guards, and the corresponding terminal paths in `operations/retry.rs` and `operations/archive.rs`. + +Strongest counterevidence: the main successful-run path is explicit and strongly partitioned, and `WorkflowLifecycle` plus `RunServices` give many responsibilities named owners. + +Why adjacent scores do not fit: 3 understates the issue because terminal completion is a central lifecycle concern, not an edge-only exception; an ordinary terminal-contract change must inspect several authorities. 1 does not fit because the normal path and the exceptional paths are still traceable and deliberately named. + +### Simplicity — 2, High confidence + +The top-level flow is readable, but routine run startup crosses a large amount of central wiring. `operations/start.rs:start` enters `execute_persisted_run`, constructs `RunSession`, and then `RunSession::run` coordinates logging, SHA listeners, initialization, cleanup/drain guards, execution, finalization, and pull-request handling. `pipeline/types.rs:InitOptions` carries a large set of run inputs, and `operations/start.rs:RunSession::run` assembles them before handing control to `pipeline/initialize.rs`. The resulting services are then repartitioned through `services.rs:RunServices`, `services.rs:EngineServices`, and `pipeline/execute.rs:execute`. + +A representative routine change is adding a run-scoped service needed by node handlers. It would pass through `operations/start.rs:StartServices` or `RunSession`, `pipeline/types.rs:InitOptions`, `pipeline/initialize.rs:initialize`, `pipeline/types.rs:Initialized`, `services.rs:RunServices`, `services.rs:EngineServices`, and the destructuring/building in `pipeline/execute.rs:execute`. + +Strongest counterevidence: the phase result types in `pipeline/types.rs` and the extracted executor/lifecycle adapters make the long path navigable; the complexity is structured rather than accidental. + +Why adjacent scores do not fit: 3 does not fit because the pressure is on the common startup and execution path, and a small run-scoped dependency change propagates through several central handoff types. 1 does not fit because the ordered pipeline and named handoffs still provide a stable path through the component. + +### Domain model — 2, High confidence + +The strongest positive mechanism is the phase model in `pipeline/types.rs`: `Parsed`, `Transformed`, `Validated`, `Persisted`, `Initialized`, `Executed`, `Concluded`, and `Finalized` constrain which data exists at each stage. Canonical run records are reused from `fabro-types`, and `services.rs:RunServices` documents cancellation ownership. + +However, the core event path weakens those guarantees. `event/events.rs:Event::StageCompleted` carries `status: String`; lifecycle code such as `lifecycle/event.rs` converts `StageOutcome` to a string, and `event/convert.rs:stage_status_from_string` parses it back when creating the durable event. An unknown value is not rejected: it is warned about and converted to `StageOutcome::Failed`. The durable model in `fabro-types` is typed, but the internal central event model permits invalid status values and gives them a lossy fallback meaning. `WorkflowRunCompleted` similarly carries a string status internally. + +Strongest counterevidence: the durable event body and most run/pipeline records use named enums and phase-specific types, so this is not a component with generally unmodeled state. + +Why adjacent scores do not fit: 3 does not fit because stage and run outcomes are central workflow vocabulary used on every execution, and the internal-to-durable boundary permits and silently reinterprets invalid values. 1 does not fit because canonical typed outcomes exist and dominate downstream storage; the break is concentrated at the internal event boundary. + +### Duplication of knowledge — 2, High confidence + +Adding an event requires coordinated knowledge in several central authorities. The internal variant lives in `event/events.rs:Event`; its wire name is separately selected by `event/names.rs:event_name`; durable fields are declared in `fabro-types::EventBody`; conversion is implemented in `event/convert.rs:event_body_from_event`; stored-field behavior is selected in `event/stored_fields.rs:stored_event_fields_for_variant`; and tracing behavior is implemented on `Event`. `docs/internal/events-strategy.md` documents this multi-site procedure, confirming that this is the expected recurring event-evolution path rather than a one-off remnant. + +A representative routine change is adding a persisted workflow event. It touches `event/events.rs:Event`, `event/names.rs:event_name`, the `Event` tracing method, `fabro_types::EventBody`, `event/convert.rs:event_body_from_event`, `event/stored_fields.rs:stored_event_fields_for_variant`, emitters, and any event consumers. + +Strongest counterevidence: `event/emitter.rs:Emitter::emit_with_scope` constructs the canonical run event once before dispatch, exhaustive matches make omissions visible to the compiler, and the strategy document gives maintainers one checklist. + +Why adjacent scores do not fit: 3 does not fit because event evolution is frequent, central workflow work and requires synchronized changes across representations and crates. 1 does not fit because each representation has a stated role and there is a single canonicalization point before dispatch. + +Lens-boundary note: the internal `Event`/durable `EventBody` split could be described as a domain-model issue or duplication. I treated the repeated declarations and conversion sites as duplication of knowledge; the separate `String`-to-`StageOutcome` loss of meaning is the domain-model issue. Likewise, repeated terminal constructors are secondary duplication, but I classified the primary problem as ownership because the key question is which operation owns terminal lifecycle completion. + +## `fabro-http` + +### Ownership boundaries — 4, High confidence + +`lib/foundation/fabro-http/src/lib.rs` is a small, focused owner for HTTP client construction and proxy policy. Callers get approved async or blocking builders and convenience clients from this crate. Repository lint policy in `clippy.toml` disallows direct `reqwest` constructors and points callers to `fabro-http`, so the boundary is reinforced rather than merely conventional. `ProxyPolicy::resolve` also owns the environment-variable authority through `fabro_static::EnvVars::FABRO_HTTP_PROXY_POLICY`. + +Strongest counterevidence: the crate deliberately re-exports several `reqwest` types and carries lint exceptions for those facade exports, so callers are not isolated from every transport detail. + +Why adjacent scores do not fit: 3 does not fit because construction policy, environment precedence, test defaults, and transport facade all have one enforced home with no observed competing builder authority. + +### Simplicity — 4, High confidence + +The common path is short: choose `HttpClientBuilder` or `BlockingHttpClientBuilder`, optionally configure it, resolve `ProxyPolicy`, and build the underlying client. `define_builder!` generates the shared async/blocking surface once, while the async-only `read_timeout` extension remains plainly visible next to the macro invocation. Convenience functions such as `http_client`, `blocking_http_client`, `test_http_client`, and `blocking_test_http_client` expose the common cases directly. + +Strongest counterevidence: macro generation means the two concrete builder implementations are not visible as ordinary source, and async-only options must be added outside the shared definition. + +Why adjacent scores do not fit: 3 does not fit because the macro removes rather than creates routine common-option work: a shared builder option is added in one readable location, while the generated types remain thin wrappers. + +### Domain model — 4, High confidence + +`ProxyPolicy` names the only supported policies, `ProxyPolicy::parse` rejects unknown values, and `ProxyPolicy::resolve_with_env_value` makes precedence explicit: a caller override wins, then the environment value, then the system default. Test helpers force `Disabled`, making local test semantics deliberate. `HttpClientBuildError` distinguishes policy configuration failure from transport construction failure. + +Strongest counterevidence: callers can express no-proxy behavior through both `proxy_policy(ProxyPolicy::Disabled)` and the lower-level `no_proxy()` builder method, and the facade re-exports lower-level proxy types. + +Why adjacent scores do not fit: 3 does not fit because the overlapping entry points do not introduce an ambiguous stored state or silent fallback: the policy values and their precedence are explicit, and invalid environment vocabulary fails closed. + +### Duplication of knowledge — 3, High confidence + +The builder macro is a strong anti-duplication mechanism for async and blocking clients. The remaining policy vocabulary is manually repeated: `ProxyPolicy` variants, `ProxyPolicy::parse`, the expected-value text in `HttpClientBuildError::InvalidProxyPolicy`, and the policy match in the generated `build` method must agree. + +A representative routine change is adding another supported proxy policy. It would touch `ProxyPolicy`, `ProxyPolicy::parse`, the expected-value message on `HttpClientBuildError::InvalidProxyPolicy`, the `define_builder!` build-time match, and policy tests in the same source file. + +Strongest counterevidence: every repeated policy decision is co-located in one small file, and the exhaustive build match makes a missing behavioral branch a compile error. + +Why adjacent scores do not fit: 4 does not fit because the accepted vocabulary and error vocabulary are independently maintained strings. 2 does not fit because the synchronization is confined to one authority and does not force routine callers or neighboring components to change. + +Lens-boundary note: macro use could be counted as simplicity indirection, but its primary effect here is eliminating async/blocking duplication. The generated control flow is small enough that I did not lower simplicity for it. + +## `fabro-web-app` + +### Ownership boundaries — 4, High confidence + +The app has explicit composition points. `app/entry.tsx` selects normal or install mode and installs shared providers; `app/router.tsx` and `app/install-router.tsx` own the two route trees. `app/lib/api-client.ts` owns generated-client construction and uniform API errors, `app/lib/query-keys.ts` owns cache keys, and `app/lib/queries.ts` owns shared reads. The React effects policy is embodied by approved wrappers in `app/hooks/effects.ts`; direct effect usage is concentrated in hooks and live-event libraries rather than route/component bodies. `scripts/build.ts` separately owns deterministic asset building and atomic publication. + +Strongest counterevidence: some cache mutation and API-write coordination remains in route handlers, particularly in the large run and installation screens, so not every server interaction passes through a single application-service layer. + +Why adjacent scores do not fit: 3 does not fit because routing, reads, client configuration, effects, and build publication each have a visible and consistently used owner; route-local writes are appropriate UI orchestration rather than a competing global authority. + +### Simplicity — 2, High confidence + +The normal routing shell is simple, but two central screens concentrate substantial policy and presentation. `app/routes/run-stages.tsx` combines event-to-turn reduction, event filtering, grouping, stage/activity interpretation, row and panel rendering, stage renderer selection, and the route page. `app/install-app.tsx` similarly combines installation state transitions, controller behavior, forms, and view composition. Cross-tab stream coordination in `app/lib/cross-tab-sse.ts` is another large central mechanism. + +A representative routine change is showing a new kind of stage activity in the run timeline. It requires following `app/lib/run-events.ts:STAGE_ACTIVITY_EVENT_TYPES`, `app/routes/run-stages.tsx:STAGE_ACTIVITY_EVENT_SET`, `app/routes/run-stages.tsx:buildStageActivity`, the route's turn/activity types, and the corresponding render helpers in the same large route module. + +Strongest counterevidence: shared event lists, query keys, generated API types, and route helpers provide landmarks, and the activity reducer is deterministic rather than dispersed among many components. + +Why adjacent scores do not fit: 3 does not fit because run-stage interpretation is a common product path and small presentation changes require navigating large modules that mix reduction and rendering concerns. 1 does not fit because the route and install flows remain typed, testable, and traceable from explicit entry points. + +### Domain model — 2, High confidence + +Generated API types provide a strong canonical model for ordinary request/response queries, and several local models use discriminated unions. The live-event boundary is weaker. `app/lib/sse.ts:EventPayload` permits an optional event name plus arbitrary fields. `app/lib/run-events.ts:RunEventPayload` and `app/lib/live-events.ts:LiveEventPayload` repeat mostly optional envelope fields with `properties: unknown`. `app/lib/sse.ts:subscribeToSharedEventSource` parses JSON and casts it to the requested payload type without runtime validation. Common live UI behavior therefore accepts payloads that lack the fields implied by their event names. + +There is additional vocabulary translation in `app/data/runs.ts:RunStatus`, which locally reproduces API run-state kinds and adds presentation state, and compatibility shape probing in `app/lib/run-sandbox-lifecycle.ts:sandboxLifecycleKind` and `sandboxInstance`. + +Strongest counterevidence: generated types remain the authority for normal API calls, `session-stream.ts` and query paths use generated event-envelope types where possible, and the local run status adds a genuine presentation concept rather than merely renaming every API state. + +Why adjacent scores do not fit: 3 does not fit because SSE drives common live run behavior and its central payload model makes invalid event/field combinations representable and unchecked. 1 does not fit because static generated models are sound and the weak representation is concentrated at live and compatibility boundaries. + +### Duplication of knowledge — 2, High confidence + +Live refresh policy is repeated in separate manually curated authorities. `app/lib/run-events.ts:RUN_SUMMARY_EVENTS` lists events that invalidate run summaries, while `app/lib/board-events.ts:BOARD_STATUS_EVENTS` independently lists many of the same run, interview, and pull-request lifecycle events for board refresh. The duplicated payload interfaces in `run-events.ts` and `live-events.ts` add another synchronization surface. + +A representative routine change is adding a lifecycle event that changes both a run summary and its board status. It requires updating `app/lib/run-events.ts:RUN_SUMMARY_EVENTS` and `app/lib/board-events.ts:BOARD_STATUS_EVENTS`, then checking phase derivation in `app/lib/run-phases.ts:deriveRunPhases` and live consumers if the event also changes the visible run phase. + +Strongest counterevidence: stage activity vocabulary is centralized in `app/lib/run-events.ts:STAGE_ACTIVITY_EVENT_TYPES` and imported by the run-stages route; query keys and server contract types are also centralized or generated. + +Why adjacent scores do not fit: 3 does not fit because the repeated invalidation lists govern common live behavior, and a missing update produces stale UI rather than a compile-time failure. 1 does not fit because each list has a clear local purpose and several other high-change vocabularies already have a single authority. + +Lens-boundary note: the repeated loose live-event interfaces are both duplicate declarations and a weak model. I treated representable invalid payloads and unchecked casts as the domain-model finding; I used independently maintained event-invalidation sets as the primary duplication finding. The size of `run-stages.tsx` is primarily simplicity pressure, not evidence that its route ownership is unclear. + +## `repository-ci` + +### Ownership boundaries — 4, High confidence + +`.github/workflows/rust.yml` and `.github/workflows/typescript.yml` have an explicit language split and named jobs for formatting, linting, generated documentation, tests, type checking, and builds. Each workflow sets narrow permissions, concurrency behavior is visible, and toolchain/action versions are pinned. The TypeScript build job's Rust build step has a clear purpose: verify the embedded production SPA through the repository's actual build command. + +Strongest counterevidence: the Rust clippy job contains a repository-specific legacy-auth `git grep` policy check, rather than delegating that policy to a named script or dedicated job. + +Why adjacent scores do not fit: 3 does not fit because the special check is still plainly owned by repository validation, while language-level checks, permissions, and production build validation have unambiguous homes and no competing workflow was observed. + +### Simplicity — 3, High confidence + +The workflows are short and linear, with direct commands corresponding to local development commands. Friction is isolated: setup steps are repeated across jobs, the clippy job embeds a multi-pattern shell assertion for legacy auth identity removal, and the ignored twin E2E selection is encoded directly in a long `nextest` expression. These cost attention but do not obscure the overall validation flow. + +A representative routine change is adding a new TypeScript validation job. It would repeat the checkout, Bun setup, and dependency-install sequence already present in `.github/workflows/typescript.yml:jobs.typecheck`, `jobs.test`, and `jobs.build`, then add the new command. + +Strongest counterevidence: each job can be understood independently, commands are explicit, and there is no multi-layer reusable-workflow indirection. + +Why adjacent scores do not fit: 4 does not fit because repeated setup and inline special policies add avoidable local friction. 2 does not fit because ordinary check changes still have a direct path through one small workflow and do not cross a complex control structure. + +### Domain model — 2, High confidence + +Some configuration identifiers no longer denote repository reality. Both push and pull-request triggers in `.github/workflows/rust.yml` refer to `openapi/**`, but that path does not exist; the actual API contract is `docs/public/api-reference/fabro-api.yaml`, which the same workflow's legacy-auth check names directly. `.github/workflows/typescript.yml` also omits that contract path even though the TypeScript API client is generated from it. A contract-only change can therefore fall outside the configured validation vocabulary. + +`.github/zizmor.yml:rules.stale-action-refs.ignore` identifies three exceptions by `rust.yml` source line. History shows those locations originally denoted Rust toolchain actions, while the current line numbers point elsewhere after workflow edits. The exception's identity is coupled to incidental layout rather than the action it is meant to describe. + +Strongest counterevidence: jobs, test modes, toolchain versions, permissions, and build profiles are otherwise named explicitly and line up with repository commands. + +Why adjacent scores do not fit: 3 does not fit because the stale/nonexistent identifiers affect whether central source-of-truth changes are validated and whether static-validation exceptions retain their intended meaning. 1 does not fit because most CI vocabulary remains stable and the affected values can be corrected from clear repository authorities. + +### Duplication of knowledge — 2, High confidence + +Trigger-path knowledge is repeated in every workflow and twice within each workflow: `.github/workflows/rust.yml:on.push.paths` duplicates `on.pull_request.paths`, and `.github/workflows/typescript.yml` does the same. Cross-language contract inputs then require synchronized edits in both files. The stale `openapi/**` entry and omission of `docs/public/api-reference/fabro-api.yaml` are direct evidence that this repeated knowledge has drifted. + +A representative routine change is moving or adding a source-of-truth file that must trigger all relevant CI. It requires updating `rust.yml:on.push.paths`, `rust.yml:on.pull_request.paths`, `typescript.yml:on.push.paths`, and `typescript.yml:on.pull_request.paths`; there is no shared authority that makes one update cover the four consumers. + +Strongest counterevidence: commands and action versions are local to their jobs, so much of the visible repetition is deliberate job isolation, and each language workflow is small. + +Why adjacent scores do not fit: 3 does not fit because trigger selection is central to CI's purpose, the synchronization crosses both event sections and language workflows, and actual drift is present. 1 does not fit because the duplicated lists are easy to locate and most entries still agree. + +Lens-boundary note: the stale OpenAPI trigger could be scored only as duplicate path knowledge. I used the repeated four-list maintenance burden for duplication, while treating the fact that `openapi/**` currently has no referent—and that line-based Zizmor identities no longer name the intended actions—as domain vocabulary drift. diff --git a/.chisel/calibration/work/reviewer-2.md b/.chisel/calibration/work/reviewer-2.md new file mode 100644 index 000000000..a9082f5c0 --- /dev/null +++ b/.chisel/calibration/work/reviewer-2.md @@ -0,0 +1,176 @@ +# Calibration Sample Review — Reviewer 2 + +Revision: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +This review uses the component boundaries in `.chisel/cartography/codebase-map.md`. In particular, `fabro-web-app` excludes `apps/fabro-web/app/components/playground/**`, and `repository-ci` contains only `.github/workflows/rust.yml`, `.github/workflows/typescript.yml`, and `.github/zizmor.yml`. + +## Score summary + +| Component | Ownership and boundaries | Simplicity | Domain model | Duplication of knowledge | +|---|---:|---:|---:|---:| +| `fabro-workflow` | 3 (Medium) | 2 (High) | 3 (Medium) | 2 (High) | +| `fabro-http` | 4 (High) | 4 (High) | 3 (High) | 4 (High) | +| `fabro-web-app` | 4 (Medium) | 2 (Medium) | 2 (Medium) | 2 (Medium) | +| `repository-ci` | 4 (High) | 3 (High) | 3 (High) | 2 (High) | + +## `fabro-workflow` + +### `ownership-boundaries` — 3, Medium confidence + +The component has a recognizable high-level owner and intended dependency direction. `lib/components/fabro-workflow/src/operations/mod.rs` owns run-level operations, while `lib/components/fabro-workflow/src/pipeline/mod.rs` owns the ordered phase API. `lib/components/fabro-workflow/src/pipeline/types.rs:Parsed`, `Transformed`, `Validated`, `Persisted`, `Initialized`, `Executed`, `Concluded`, and `Finalized` make phase ownership explicit. `lib/components/fabro-workflow/src/services.rs:RunServices` and `EngineServices` distinguish run-lifetime services from node-execution services, and `lib/components/fabro-workflow/src/node_handler.rs:WorkflowNodeHandler` is a visible adapter to `fabro-core`. + +The friction is at the public edge: `lib/components/fabro-workflow/src/lib.rs` exposes operations, pipeline phases, handlers, records, services, runtime storage, and several `#[doc(hidden)]` modules. Callers can therefore enter below the complete lifecycle as well as through `lib/components/fabro-workflow/src/operations/start.rs:start`. This weakens containment, but it does not create a competing production owner. + +**Strongest counterevidence:** The typed phase outputs and the `RunServices`/`EngineServices` split strongly reinforce one workflow lifecycle. + +**Why adjacent scores do not fit:** A 4 does not fit because the broad facade exposes enough lifecycle internals to make the boundary porous. A 2 does not fit because the normal `start` path and each phase owner remain identifiable and dependencies are delegated to dedicated crates. + +### `simplicity` — 2, High confidence + +The stable common path is traceable, but routine work crosses substantial central machinery: `lib/components/fabro-workflow/src/operations/start.rs:start` → `execute_persisted_run` → `RunSession::new` → `RunSession::run` → `pipeline::initialize` → `pipeline::execute` → `pipeline::finalize` → `pipeline::pull_request`. Along that path, `StartServices`, `RunSession`, and `lib/components/fabro-workflow/src/pipeline/types.rs:InitOptions` each carry many run concerns, while bootstrap, completion, cleanup, steering-drain, sandbox, and event-flush guards add multiple exit paths. `lib/components/fabro-workflow/src/pipeline/initialize.rs:initialize` also coordinates sandbox creation/reconnection, hooks, credentials, Git setup, handler construction, and resume state. + +**Representative routine change:** Adding one run-scoped execution service would normally thread through `operations/start.rs:StartServices`, `RunSession`, and `RunSession::new`; `pipeline/types.rs:InitOptions`; `pipeline/initialize.rs:initialize`; and `services.rs:RunServices` or `EngineServices`. + +**Strongest counterevidence:** `operations/start.rs:RunSession::run` presents the main phases in a linear order, and the phase-specific types preserve that order despite the setup machinery. + +**Why adjacent scores do not fit:** A 3 does not fit because the pressure is on the main run path rather than at an edge. A 1 does not fit because there is a stable phase sequence and named service bundles to follow. + +### `domain-model` — 3, Medium confidence + +The strongest mechanism is the phase-state model in `lib/components/fabro-workflow/src/pipeline/types.rs`; private fields on `Validated` and `Persisted` and opaque `ResumeState` prevent several invalid transitions. `lib/components/fabro-workflow/src/pipeline/finalize.rs:classify_engine_result` is also a clear authority for translating an engine result into `StageOutcome`, failure detail, and `RunStatus`. + +The main friction is the extensible, string-valued handler vocabulary on the common graph path. `lib/components/fabro-workflow/src/handler/mod.rs:HandlerRegistry::resolve` works with type strings and falls back to the default handler, while `default_registry` registers the built-in strings. Validation in `fabro-validate` protects normal runs, but execution itself does not carry a closed built-in handler type. + +**Strongest counterevidence:** `pipeline/types.rs:ResumeState::from_projection`, the phase output types, and `pipeline/finalize.rs:classify_engine_result` give important workflow concepts one enforced shape. + +**Why adjacent scores do not fit:** A 4 does not fit because handler identity remains string-valued and default-resolved through a central execution boundary. A 2 does not fit because validation and typed phase states canonicalize the normal run before execution. + +### `duplication-knowledge` — 2, High confidence + +Event knowledge is repeated across central authorities. `lib/components/fabro-workflow/src/event/events.rs:Event` defines the emitter-facing shape, `lib/components/fabro-workflow/src/event/convert.rs:event_body_from_event` translates it to the stored `fabro_types::EventBody`, `lib/components/fabro-workflow/src/event/names.rs:event_name` separately assigns wire names, and `lib/components/fabro-workflow/src/event/stored_fields.rs:stored_event_fields_for_variant` separately assigns envelope metadata. These exhaustive matches help detect omissions, but every ordinary event extension still requires synchronized semantic decisions. + +**Representative routine change:** Adding a stored workflow event can touch `event/events.rs:Event`, `event/convert.rs:event_body_from_event`, `event/names.rs:event_name`, `event/stored_fields.rs:stored_event_fields_for_variant`, and the canonical `lib/foundation/fabro-types/src/run_event/mod.rs:EventBody` authority. + +**Strongest counterevidence:** `event/convert.rs:to_run_event_at` is the single assembly point, and Rust's exhaustive matches turn many missed updates into compile failures. + +**Why adjacent scores do not fit:** A 3 does not fit because event emission and persistence are central, recurring behavior. A 1 does not fit because the authorities are explicit and compiler-checked rather than unidentifiable. + +## `fabro-http` + +### `ownership-boundaries` — 4, High confidence + +`lib/foundation/fabro-http/src/lib.rs` has one focused transport-construction boundary. `HttpClientBuilder`, `BlockingHttpClientBuilder`, `ProxyPolicy`, the client aliases, and the production/test constructors all live there; the crate depends only on `fabro-static`, `reqwest`, and `thiserror`. Repository policy reinforces the boundary through `clippy.toml:disallowed-methods`, which directs raw reqwest construction to this facade. + +**Strongest counterevidence:** The public reqwest aliases and re-exports make the abstraction intentionally permeable, so it does not own higher-level request behavior. + +**Why the adjacent score does not fit:** A 3 does not fit because exposing reqwest types is part of the mapped purpose, while construction policy and proxy resolution still have one clear owner. + +### `simplicity` — 4, High confidence + +`lib/foundation/fabro-http/src/lib.rs:define_builder` expresses shared async/blocking forwarding once. Both builders end at the same short `ProxyPolicy::resolve` and `build` path, and `http_client`, `test_http_client`, `blocking_http_client`, and `blocking_test_http_client` are thin named entry points. A shared reqwest builder option is normally added once to the macro. + +**Strongest counterevidence:** The macro hides generated methods, and async-only `HttpClientBuilder::read_timeout` must sit outside it. + +**Why the adjacent score does not fit:** A 3 does not fit because this indirection directly removes twin implementations and leaves callers with a single conventional builder path. + +### `domain-model` — 3, High confidence + +`lib/foundation/fabro-http/src/lib.rs:ProxyPolicy` gives the repository policy two named states, `ProxyPolicy::resolve_with_env_value` defines explicit-over-environment precedence, and `HttpClientBuildError::InvalidProxyPolicy` rejects unknown values. The tests cover default, environment, invalid, and explicit-override cases. + +The isolated ambiguity is that `HttpClientBuilder::no_proxy` and `HttpClientBuilder::proxy_policy(ProxyPolicy::Disabled)` both publicly express disabled proxy behavior, but `no_proxy` mutates the inner builder without updating the policy field. Their relationship is not represented or documented in the type. + +**Strongest counterevidence:** The closed enum, typed error, and resolver tests make the environment-facing policy meaning unusually explicit. + +**Why adjacent scores do not fit:** A 4 does not fit because two public controls overlap without an encoded relationship. A 2 does not fit because the overlap is local and every normal constructor still passes through one two-state resolver. + +### `duplication-knowledge` — 4, High confidence + +The builder macro is the authority for behavior shared by synchronous and asynchronous clients, and every constructor delegates to those builders. The production/test and async/blocking helper names repeat syntax, not policy: test behavior is expressed once as `ProxyPolicy::Disabled`. + +**Strongest counterevidence:** Four constructor helpers and the separate async-only impl are superficially repetitive. + +**Why the adjacent score does not fit:** A 3 does not fit because changing proxy precedence or disabled behavior has one authority; the remaining repetition does not require synchronized policy decisions. + +## `fabro-web-app` + +### `ownership-boundaries` — 4, Medium confidence + +The main browser lifecycle has clear homes. `apps/fabro-web/app/entry.tsx` selects install or normal routing and owns root providers; `app/router.tsx:routes` owns the product route graph; `app/install-router.tsx:installRoutes` owns first-run routing; `app/lib/api-client.ts` owns HTTP normalization; `app/lib/queries.ts` and `app/lib/mutations.ts` own shared server access; and `app/hooks/effects.ts` contains reusable browser-effect lifecycles. Route modules own page-specific composition. The separately mapped playground enters through `app/router.tsx` without its excluded implementation being absorbed into this assessment. + +**Strongest counterevidence:** `app/routes/run-stages.tsx` and `app/install-app.tsx` each combine page state, domain projection, and rendering in one route-owned file. + +**Why the adjacent score does not fit:** A 3 does not fit because those combinations create local complexity, but no competing owner or reversed dependency was identified; shared cross-route responsibilities still have clear modules. + +### `simplicity` — 2, Medium confidence + +Two common product paths carry central transformation machinery. `apps/fabro-web/app/routes/run-stages.tsx` turns event envelopes into `TurnType` values in `buildStageActivity`, then separately groups, filters, timelines, labels, summarizes, and renders them through `buildChatItems`, `groupConsecutiveTools`, `filterDisplayItems`, `buildThreadDnaItems`, and the route's view components. `apps/fabro-web/app/install-app.tsx` similarly contains the install reducer, session hydration, controller, step forms, review, finishing, payload construction, and supporting controls in one flow. + +**Representative routine change:** Changing how a tool event appears on the stage page requires tracing `run-stages.tsx:buildStageActivity`, `buildChatItems`/`groupConsecutiveTools`, `buildThreadDnaItems`, `turnLabel`, `turnSummary`, `EventDetails`, and `StageChatView`. + +**Strongest counterevidence:** The stage path uses discriminated unions and mostly pure exported transformations with focused tests, so each individual step can be reasoned about. + +**Why adjacent scores do not fit:** A 3 does not fit because the long transformation chains are central to major routes. A 1 does not fit because the named pure functions provide a stable trace through both flows. + +### `domain-model` — 2, Medium confidence + +Generated API types provide a useful boundary, but the central event path accepts several simultaneous shapes. `apps/fabro-web/app/lib/run-events.ts:RunEventPayload` makes event identity and metadata optional and `stageIdFromPayload` falls back from `stage_id` to `node_id` to `properties.node_id`. `app/routes/run-stages.tsx:activityEventStageId` repeats that shape tolerance for stored `EventEnvelope`s, while `buildStageActivity` reads tool, text, argument, and output values from both `properties` and legacy top-level fields via `app/lib/unknown.ts`. + +**Representative routine change:** Moving one stage-event field to its canonical envelope location can require coordinated interpretation changes in `lib/run-events.ts:RunEventPayload` and `stageIdFromPayload`, plus `routes/run-stages.tsx:activityEventStageId` and `buildStageActivity`. + +**Strongest counterevidence:** Once parsed, `run-stages.tsx:TurnType`, `StageRenderer`, and generated `StageHandler`/`StageState` types give the UI clear closed shapes. + +**Why adjacent scores do not fit:** A 3 does not fit because the multi-shape event interpretation is on live invalidation and the main stage view, not an edge. A 1 does not fit because generated types and discriminated UI projections establish a stable canonical shape after parsing. + +### `duplication-knowledge` — 2, Medium confidence + +Stage-state presentation policy is authoritative in several common views. `apps/fabro-web/app/lib/stage-sidebar.ts:ACTIVE_STAGE_STATES`, `IN_FLIGHT_STAGE_STATES`, `SUCCEEDED_STAGE_STATES`, `STAGE_STATUS_TONE`, and `STAGE_STATUS_LABEL` define classifications and visuals, while `app/components/stage-sidebar.tsx:statusConfig`, `app/components/run-waterfall.tsx:stageBarClass` and `isStageInFlight`, and `app/components/stage-popover.tsx:StatusPill` make parallel state decisions. + +**Representative routine change:** Adding a generated `StageState` requires reviewing or changing all of those authorities so the sidebar, waterfall, and popover agree on activity, success, label, and tone. + +**Strongest counterevidence:** Generated `StageState` plus exhaustive `Record` mappings catch many omissions, and `lib/stage-sidebar.ts` already centralizes several shared classifications. + +**Why adjacent scores do not fit:** A 3 does not fit because stage status is central to multiple routine run views and synchronization is recurring. A 1 does not fit because the generated enum is a clear semantic authority and TypeScript catches many missing cases. + +## `repository-ci` + +### `ownership-boundaries` — 4, High confidence + +The two workflows divide validation by ecosystem: `.github/workflows/rust.yml:jobs` owns Rust format, lint, generated-doc, workspace test, twin-mode ignored tests, and manual macOS validation; `.github/workflows/typescript.yml:jobs` owns web/client typecheck, web tests, and the embedded-SPA production build. Both use top-level empty permissions and job-local read permission. The cross-language Cargo build in the TypeScript build job validates the mapped embedded-SPA integration rather than creating a second build owner. + +**Strongest counterevidence:** The Rust clippy job contains a repository-wide legacy-auth guard that also scans TypeScript and API paths. + +**Why the adjacent score does not fit:** A 3 does not fit because that cross-language invariant remains an explicitly named CI check, while job and workflow lifecycle ownership stays clear. + +### `simplicity` — 3, High confidence + +The main flow is explicit: named jobs perform checkout, tool setup, and one or two direct repository commands. The isolated friction is `.github/workflows/rust.yml:jobs.clippy.steps.Verify legacy auth identity removal`, where a long regular expression and shell exit-status protocol are embedded in a lint job. The twin-mode test semantics also need a substantial comment and package expression in `jobs.test`. + +**Strongest counterevidence:** Separate jobs, direct commands, pinned tools, and no reusable-workflow indirection make routine CI behavior easy to locate. + +**Why adjacent scores do not fit:** A 4 does not fit because the legacy guard and twin-mode selection require non-obvious local interpretation. A 2 does not fit because that machinery is isolated and ordinary check changes still follow a direct job structure. + +### `domain-model` — 3, High confidence + +Job names, triggers, permissions, platforms, and commands have consistent meanings in the GitHub Actions structure. Exact action SHAs and named modes such as `--profile ci` reduce ambiguity. The main gap is that `.github/workflows/rust.yml:jobs.test` relies on the external default meaning of `FABRO_TEST_MODE` for its twin run rather than setting the mode in the workflow; the comment is the only local declaration of that state. + +**Strongest counterevidence:** The command, package selector, and explanation tightly describe the intended twin-only behavior, and every job has an explicit runner and permission set. + +**Why adjacent scores do not fit:** A 4 does not fit because a central test mode is implicit in an external default. A 2 does not fit because the rest of the workflow vocabulary is coherent and the implicit state is limited to one documented test step. + +### `duplication-knowledge` — 2, High confidence + +Trigger policy is repeated verbatim between `on.push.paths` and `on.pull_request.paths` in both workflow files. Action versions and bootstrap steps are also copied across every job. `.github/zizmor.yml:rules.stale-action-refs.ignore` adds line-number references to `rust.yml`, creating another manually synchronized representation; at this revision its listed lines 37, 49, and 62 are respectively a blank line, the `fmt` job key, and a Cargo command rather than action references. + +**Representative routine change:** Adding a new Rust-owned source area requires matching edits to `.github/workflows/rust.yml:on.push.paths` and `on.pull_request.paths`; upgrading checkout requires synchronized edits in `jobs.fmt`, `clippy`, `generated-docs`, `test`, and `test-macos`, followed by review of `.github/zizmor.yml:rules.stale-action-refs.ignore`. + +**Strongest counterevidence:** The duplication is explicit and small enough to inspect, and each actual validation command appears once in its intended job. + +**Why adjacent scores do not fit:** A 3 does not fit because triggers and action versions are central, recurring maintenance knowledge and the stale line selectors demonstrate drift. A 1 does not fit because the canonical workflows and intended checks remain identifiable. + +## Lens-boundary confusion + +- The `fabro-workflow` `Event`/`EventBody` split could be described as two domain shapes. I assigned its score effect to `duplication-knowledge` because the discriminating problem is the synchronized event name, conversion, and envelope-field decisions, not an inability to identify either type's meaning. +- The size and mixed contents of `fabro-web-app` route files could look like misplaced responsibility. I assigned the main effect to `simplicity` because the route remains the clear owner; the problem is tracing the amount of local machinery. +- Repeated `StageState` maps could be treated as domain drift. I assigned them to `duplication-knowledge` because the generated enum preserves meaning and the observed burden is repeating presentation/classification policy across views. +- The `.github/zizmor.yml` line selectors could be treated as invalid configuration meaning. I assigned their main effect to `duplication-knowledge` because the failure mechanism is manual synchronization with line positions; `repository-ci` domain scoring instead uses the implicit twin-mode default. +- `fabro-http`'s macro could be treated as simplicity indirection, while its two proxy-disable controls could be treated as duplicate policy. I treated the macro as a positive simplicity/duplication mechanism and the overlapping controls as `domain-model` friction because the unresolved question is what each public control means. diff --git a/.chisel/calibration/work/reviewer-3.md b/.chisel/calibration/work/reviewer-3.md new file mode 100644 index 000000000..98558bff4 --- /dev/null +++ b/.chisel/calibration/work/reviewer-3.md @@ -0,0 +1,490 @@ +# Calibration Sample Review — Reviewer 3 + +Revision: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +Scope follows `.chisel/cartography/codebase-map.md`: `fabro-workflow`, +`fabro-http`, `fabro-web-app`, and `repository-ci`. The `fabro-web-app` +reading excludes `apps/fabro-web/app/components/playground/**`; +`repository-ci` includes only `.github/workflows/rust.yml`, +`.github/workflows/typescript.yml`, and `.github/zizmor.yml`. + +## Provisional Matrix + +| Component | Ownership and boundaries | Simplicity | Domain model | Duplication of knowledge | +|---|---:|---:|---:|---:| +| `fabro-workflow` | 4 / High | 2 / High | 2 / High | 2 / High | +| `fabro-http` | 4 / High | 4 / High | 4 / High | 4 / High | +| `fabro-web-app` | 3 / High | 2 / High | 2 / High | 2 / High | +| `repository-ci` | 3 / High | 3 / High | 2 / High | 2 / High | + +## `fabro-workflow` + +### `ownership-boundaries` — 4, High confidence + +Evidence: + +- `lib/components/fabro-workflow/src/pipeline/mod.rs` exposes an ordered phase + facade, while `pipeline/types.rs:Parsed`, `Transformed`, `Validated`, + `Persisted`, `Initialized`, `Executed`, `Concluded`, and `Finalized` give each + phase an explicit handoff. +- `lib/components/fabro-workflow/src/handler/mod.rs:Handler` and + `HandlerRegistry` own workflow-specific dispatch; + `src/node_handler.rs:WorkflowNodeHandler` is the narrow adapter to + `fabro_core::handler::NodeHandler`. +- `lib/components/fabro-workflow/src/lifecycle/mod.rs:WorkflowLifecycle` states + that it owns callback ordering and delegates event, hook, fidelity, + auto-status, circuit-breaker, Git, and artifact work to focused lifecycle + objects. +- `lib/components/fabro-workflow/Cargo.toml:[dependencies]` points from the + orchestrator to parsing, validation, sandbox, persistence, model, and generic + execution crates; generic traversal remains in `fabro-core`. + +Strongest counterevidence: startup state is carried through +`operations/start.rs:StartServices`, `RunSession`, +`pipeline/types.rs:InitOptions`, and `services.rs:RunServices` / +`EngineServices`, so the lifecycle boundary has substantial wiring. + +Why adjacent scores do not fit: 3 would treat that wiring as unclear ownership, +but the common path consistently identifies phase, handler, lifecycle, and +generic-executor owners. The counterevidence is primarily machinery inside the +intended orchestration owner, not a competing dependency direction or lifecycle +home. + +### `simplicity` — 2, High confidence + +Evidence: + +- The normal start path crosses + `operations/start.rs:start` → `execute_persisted_run` → + `RunSession::new` → `RunSession::run` → + `pipeline::initialize` → `pipeline::execute` → + `pipeline::finalize` → `pipeline::pull_request`. +- The same run-scoped collaborators are reshaped across + `operations/start.rs:StartServices`, `RunSession`, + `pipeline/types.rs:InitOptions`, `services.rs:RunServices`, and + `EngineServices`. +- `lifecycle/mod.rs:WorkflowLifecycle::new` takes the full set of lifecycle + collaborators and has an explicit `too_many_arguments` exception before + constructing seven sub-lifecycles with shared coordination state. + +Strongest counterevidence: the phase-state types in +`pipeline/types.rs` and the focused handler/lifecycle modules make this +machinery traceable; the common path is not hidden. + +Why adjacent scores do not fit: 3 does not fit because every ordinary run +traverses the service reshaping and multi-stage cleanup/finalization path; this +is central rather than edge friction. 1 does not fit because the named phase +sequence and handoff types provide a stable path through the machinery. + +Representative routine change: adding a run-scoped execution-audit sink for +handlers would require threading it through +`operations/start.rs:StartServices`, `RunSession`, +`RunSession::new`, `RunSession::run`, +`pipeline/types.rs:InitOptions`, `pipeline/initialize.rs:initialize`, and +`services.rs:RunServices` or `EngineServices`. + +### `domain-model` — 2, High confidence + +Evidence: + +- Positive mechanisms are substantial: + `pipeline/types.rs:Validated` hides its graph and exposes validation + operations, `ResumeState::from_projection` creates opaque resume state, and + `run_status.rs` plus `outcome.rs` reuse canonical types from `fabro-types` and + `fabro-core`. +- A central exception remains: + `event/events.rs:Event::StageCompleted` represents `status` as `String`, while + execution uses typed `outcome.rs:StageOutcome`. + `event/convert.rs:stage_status_from_string` reparses the string and maps every + unknown value to a failed outcome. +- The common producer + `lifecycle/event.rs:EventLifecycle::after_node` converts the typed outcome to + a string before the canonical event conversion converts it back. + +Strongest counterevidence: the pipeline phase types, `RunStatus`, +`StageOutcome`, `StageId`, and the durable `fabro_types::EventBody` otherwise +give the main workflow concepts canonical typed shapes. + +Why adjacent scores do not fit: 3 does not fit because stage completion is on +the execution hot path and accepts states the canonical outcome enum rejects. +1 does not fit because the canonical types and phase states still give the +workflow a coherent vocabulary overall. + +Representative routine change: adding or changing a stage outcome would touch +the canonical `lib/foundation/fabro-core/src/outcome.rs:StageOutcome`, string +construction in `lifecycle/event.rs:EventLifecycle::after_node`, +`event/events.rs:Event::StageCompleted`, +`event/convert.rs:stage_status_from_string`, and terminal interpretation in +`pipeline/finalize.rs:classify_engine_result`. + +### `duplication-knowledge` — 2, High confidence + +Evidence: + +- `event/events.rs:Event` defines the internal event shape, + `event/names.rs:event_name` independently maps every variant to its external + name, `event/stored_fields.rs:stored_event_fields` independently selects + envelope fields, and `event/convert.rs:event_body_from_event` constructs the + canonical `fabro_types::EventBody`. +- `docs/internal/events-strategy.md:Adding A New Event` explicitly requires + synchronized edits to the internal event, tracing, external name, + `EventBody`, stored fields, conversion, and consumers. +- Exhaustive matches make omissions visible, but they do not make one of those + mappings authoritative for the others. + +Strongest counterevidence: `event/emitter.rs:Emitter` canonicalizes each emitted +event once, all listeners receive the same `RunEvent`, and exhaustive matching +plus conversion tests detect much of the synchronization drift. + +Why adjacent scores do not fit: 3 does not fit because adding an event is a +routine extension to this component and centrally requires several independent +authorities. 1 does not fit because the events strategy clearly identifies all +authorities and the compiler/test suite gives a stable update path. + +Representative routine change: adding `run.suspended` would touch +`event/events.rs:Event`, `events.rs:Event::trace`, +`event/names.rs:event_name`, +`lib/foundation/fabro-types/src/run_event/mod.rs:EventBody`, +`event/stored_fields.rs:stored_event_fields`, +`event/convert.rs:event_body_from_event`, and relevant store/UI consumers. + +## `fabro-http` + +### `ownership-boundaries` — 4, High confidence + +Evidence: + +- The component is one focused source module: + `lib/foundation/fabro-http/src/lib.rs` owns the reqwest facade, + `ProxyPolicy`, client builders, build errors, and deterministic test clients. +- `src/lib.rs:HttpClientBuilder::build` and + `BlockingHttpClientBuilder::build` are the construction boundary where the + process proxy policy is applied. +- `clippy.toml:disallowed-methods` denies direct reqwest client constructors and + points callers to this component; `fabro_static::EnvVars` supplies the one + environment-variable name without introducing higher-level configuration. + +Strongest counterevidence: the facade deliberately re-exports many reqwest +types, and exceptional consumers still carry direct reqwest dependencies for +generated clients or incompatible dependency versions. + +Why adjacent scores do not fit: 3 does not fit because the normal async, +blocking, production, and test construction paths all converge on the same +owned policy, with a repository lint reinforcing that boundary. + +### `simplicity` — 4, High confidence + +Evidence: + +- `src/lib.rs:define_builder!` expresses the common async/blocking builder once; + the four convenience constructors are thin calls to the same builders. +- The common flow is direct: + `HttpClientBuilder::new` → optional reqwest options → + `HttpClientBuilder::build` → `ProxyPolicy::resolve` → reqwest build. +- The only async-only option is visibly isolated in + `HttpClientBuilder::read_timeout`. + +Strongest counterevidence: the macro hides the two generated impls and every +new exposed reqwest option requires another forwarding method. + +Why adjacent scores do not fit: 3 does not fit because the macro removes a real +parallel API synchronization burden while leaving the common client-building +path locally readable; its indirection is not encountered beyond this file. + +### `domain-model` — 4, High confidence + +Evidence: + +- `src/lib.rs:ProxyPolicy` has exactly the two supported states, + `ProxyPolicy::resolve_with_env_value` makes explicit configuration override + environment fallback, and invalid/non-Unicode values become + `HttpClientBuildError`. +- `src/lib.rs:HttpClientBuildError` distinguishes invalid policy from underlying + reqwest construction failure. +- `test_http_client` and `blocking_test_http_client` select the typed + `ProxyPolicy::Disabled` rather than relying on ambient test environment state. + +Strongest counterevidence: the environment boundary is necessarily stringly, +and `ProxyPolicy::parse` accepts case variants before producing the enum. + +Why adjacent scores do not fit: 3 does not fit because invalid strings are +rejected at the boundary, precedence is explicit, and all downstream paths use +the closed enum. + +### `duplication-knowledge` — 4, High confidence + +Evidence: + +- `src/lib.rs:define_builder!` is the single authority for shared async and + blocking options and policy application. +- `ProxyPolicy::resolve` is the single production authority for explicit/env/ + default precedence. +- `clippy.toml:disallowed-methods` prevents ordinary callers from silently + recreating client-construction policy outside the component. + +Strongest counterevidence: async and blocking convenience constructors remain +as four syntactically similar functions, and `read_timeout` cannot live in the +shared macro surface. + +Why adjacent scores do not fit: 3 does not fit because the remaining repetition +does not duplicate a policy or require independent decisions; it exposes +parallel entry points backed by the same authority. + +## `fabro-web-app` + +### `ownership-boundaries` — 3, High confidence + +Evidence: + +- `apps/fabro-web/app/entry.tsx:AppRuntime` owns browser bootstrap and global + runtime providers; `router.tsx:routes` and + `install-router.tsx:installRoutes` own the two route graphs. +- `app/lib/queries.ts` and `app/lib/mutations.ts` own server reads and writes; + `app/lib/api-client.ts` owns transport/error normalization. +- `app/hooks/effects.ts` and purpose-named hooks such as + `useRunEvents` and `useInstallRestartHealthPolling` contain browser resource + lifecycles rather than leaving them in route rendering. +- `routes/run-detail.tsx:RunDetail` delegates its header, actions, model, + lifecycle-toast, tab-shell, and docked-control responsibilities to the + `routes/run-detail/**` modules. + +Strongest counterevidence: two mapped common paths still concentrate several +responsibilities: +`install-app.tsx:InstallApp` / `useInstallController` contains state, +hydration, submission, step routing, payload construction, and rendering, while +`routes/run-stages.tsx:RunStages` / `buildStageActivity` contains event +interpretation and a large part of stage presentation. + +Why adjacent scores do not fit: 4 does not fit because those central route +modules are not merely edge exceptions. 2 does not fit because routes, API +access, queries, mutations, browser effects, and build lifecycle still have +stable homes and dependencies generally point through those homes. + +### `simplicity` — 2, High confidence + +Evidence: + +- The first-run common path is concentrated in + `install-app.tsx:installReducer`, `useInstallController`, `InstallApp`, + `LlmStep`, `ObjectStoreStep`, `SandboxStep`, `GithubStep`, + `buildObjectStorePayload`, and `buildSandboxPayload`. +- The run-stage common path combines + `routes/run-stages.tsx:selectStageRenderer`, + `buildStageActivity`, filtering, debug views, waterfall construction, and + `RunStages`. +- Cross-tab event sharing introduces a second substantial state machine at + `app/lib/cross-tab-sse.ts:CrossTabSseCoordinator`, beneath the already + separate shared-event-source logic in `app/lib/sse.ts:subscribeToSharedEventSource`. + +Strongest counterevidence: reducers, discriminated unions, shared query hooks, +purpose-named integration hooks, and extracted run-detail modules make many +individual flows explicit and testable. + +Why adjacent scores do not fit: 3 does not fit because installation, run-stage +inspection, and live refresh are mapped common paths, not optional edge +machinery. 1 does not fit because each path still has identifiable entry +points, state machines, and tests. + +Representative routine change: adding an installation step for telemetry would +touch `install-app.tsx:INSTALL_STEPS`, `InstallState`, `InstallAction`, +`installReducer`, `useInstallController`, `InstallApp`, a new step component, +review-summary helpers, `install-api.ts`, and the generated install API +authority in `docs/public/api-reference/fabro-api.yaml`. + +### `domain-model` — 2, High confidence + +Evidence: + +- Positive mechanisms include generated API types throughout the query and + route layers, `mode.ts:FabroMode`, and exhaustive display maps such as + `lib/sandbox-state.ts:SANDBOX_STATE_DISPLAY`. +- The central SSE boundary instead uses + `lib/sse.ts:EventPayload`, where `event` is optional and all other fields are + unknown, then extends it as + `lib/run-events.ts:RunEventPayload` with optional string identifiers and + another untyped `properties` map. +- `lib/run-events.ts:stageIdFromPayload` accepts `stage_id`, `node_id`, or + `properties.node_id` as the stage identity. +- `lib/run-sandbox-lifecycle.ts:sandboxLifecycleKind` and `sandboxInstance` + cast generated values into compatibility shapes and infer lifecycle from + either `kind`, `instance`, or legacy `runtime` / `provider` fields. + +Strongest counterevidence: normal HTTP reads and writes use +`@qltysh/fabro-api-client` types, and `Record` display maps +make many API vocabulary changes compile-visible. + +Why adjacent scores do not fit: 3 does not fit because SSE drives normal run +refresh and stage views while permitting absent event and identity fields with +multiple meanings. 1 does not fit because generated HTTP types and local +discriminated unions still provide a coherent model for most operations. + +Representative routine change: making stage identity canonical across live +events would touch the wire authority +`docs/public/api-reference/fabro-api.yaml`, +`lib/sse.ts:EventPayload`, `lib/run-events.ts:RunEventPayload`, +`stageIdFromPayload`, and consumers such as +`routes/run-stages.tsx:buildStageActivity`. + +### `duplication-knowledge` — 2, High confidence + +Evidence: + +- `lib/board-events.ts:BOARD_STATUS_EVENTS` independently decides which run + events refresh lists, while `lib/run-events.ts:RUN_SUMMARY_EVENTS`, + `TERMINAL_EVENTS`, and other sets decide detail invalidations. +- `lib/run-phases.ts:deriveRunPhases` independently matches the same lifecycle + event vocabulary to build the pre-stage timeline. +- `lib/run-events.ts:STAGE_ACTIVITY_EVENT_TYPES` is a positive local authority + shared with `routes/run-stages.tsx:buildStageActivity`, but it covers only one + slice of the broader manual event policy. + +Strongest counterevidence: list and detail invalidation are genuinely different +consumer decisions, `query-keys.ts:queryKeys` centralizes cache identities, and +the stage-activity list is deliberately shared with its reducer. + +Why adjacent scores do not fit: 3 does not fit because a normal lifecycle-event +extension that affects board and run detail requires synchronized policy edits +in separate common subscriptions. 1 does not fit because each consumer's +authority is named, localized, and covered by focused tests. + +Representative routine change: adding a `run.suspended` transition that should +refresh both list and detail views would touch +`board-events.ts:BOARD_STATUS_EVENTS`, +`run-events.ts:RUN_SUMMARY_EVENTS` (and possibly `TERMINAL_EVENTS` if its +semantics require it), `board-events.test.tsx`, `run-events.test.tsx`, and the +upstream event/OpenAPI authorities. + +## `repository-ci` + +### `ownership-boundaries` — 3, High confidence + +Evidence: + +- `.github/workflows/rust.yml:jobs` owns Rust formatting, lint, generated-doc, + Linux test, twin-E2E, and manual macOS validation. +- `.github/workflows/typescript.yml:jobs` owns browser/client typecheck, web + tests, and the embedded-SPA release build. +- Both workflows set top-level empty permissions and grant only + `contents: read` per job; all third-party actions are commit-pinned. +- Generated-document and embedded-SPA behavior is delegated to + `cargo dev docs check` and `cargo dev build`, leaving those build procedures + in `fabro-build-tooling`. + +Strongest counterevidence: +`.github/workflows/rust.yml:jobs.clippy.steps[name="Verify legacy auth identity removal"]` +contains an authentication-migration vocabulary grep inside the general CI +workflow, so an auth-domain transition also has a policy home here. + +Why adjacent scores do not fit: 4 does not fit because that product-domain +policy crosses into the CI owner and the trigger boundary has drift discussed +under domain model. 2 does not fit because the normal validation jobs and their +delegated build/test authorities remain clearly owned and directional. + +Representative routine change: renaming or restoring an authentication identity +would require changing the product types and also the legacy-name authority in +`.github/workflows/rust.yml:jobs.clippy.steps[name="Verify legacy auth identity removal"]`. + +### `simplicity` — 3, High confidence + +Evidence: + +- Each job is a short checkout/setup/command sequence, and the two workflows + split by the repository's Rust and Bun validation surfaces. +- `.github/workflows/rust.yml:jobs.test` explains the non-obvious twin-mode + expression and why it must not use the strict E2E profile. +- `.github/workflows/typescript.yml:jobs.build` delegates the mixed Rust/SPA + build to one repository command rather than reproducing its internals. + +Strongest counterevidence: checkout, tool setup, install, permissions, runner, +and cache declarations are repeated across every job; the inline legacy-auth +shell condition is more elaborate than the surrounding declarative checks. + +Why adjacent scores do not fit: 4 does not fit because routine maintenance must +scan repeated job scaffolding and one bespoke shell policy. 2 does not fit +because a contributor can still trace each common validation path directly +from one named job to one repository command. + +### `domain-model` — 2, High confidence + +Evidence: + +- `.github/workflows/rust.yml:on.push.paths` and `on.pull_request.paths` contain + `openapi/**`, but that directory does not exist at the assessed revision. +- The actual contract authority is + `docs/public/api-reference/fabro-api.yaml`, as named by + `AGENTS.md:API workflow`, + `lib/foundation/fabro-api/build.rs:main`, and + `lib/packages/fabro-api-client/package.json:scripts.generate`. +- Neither `.github/workflows/rust.yml:on.*.paths` nor + `.github/workflows/typescript.yml:on.*.paths` names that actual contract + path, even though both generated clients depend on it. + +Strongest counterevidence: job names, Rust versus TypeScript scope, twin versus +live test meaning, and toolchain versions are otherwise explicit; the commands +the jobs run correspond to checked-in project commands. + +Why adjacent scores do not fit: 3 does not fit because an ordinary edit to the +HTTP source of truth falls outside both central validation trigger models. 1 +does not fit because the workflows still have a stable and mostly accurate +vocabulary for jobs, branches, tools, and commands. + +Representative routine change: editing only +`docs/public/api-reference/fabro-api.yaml` should exercise Rust generation and +TypeScript typecheck/build, but its meaning would have to be repaired in +`.github/workflows/rust.yml:on.push.paths`, +`.github/workflows/rust.yml:on.pull_request.paths`, +`.github/workflows/typescript.yml:on.push.paths`, and +`.github/workflows/typescript.yml:on.pull_request.paths`. + +### `duplication-knowledge` — 2, High confidence + +Evidence: + +- Each workflow repeats its path set under both `on.push.paths` and + `on.pull_request.paths`; a new CI-relevant repository path has two authorities + per language. +- `.github/workflows/rust.yml:jobs.fmt`, `jobs.clippy`, + `jobs.generated-docs`, `jobs.test`, and `jobs.test-macos` independently repeat + checkout pins, credential policy, runner/toolchain setup, and often cache + setup. +- `.github/workflows/typescript.yml:jobs.typecheck`, `jobs.test`, and + `jobs.build` independently repeat checkout, Bun setup, and frozen install. + +Strongest counterevidence: independent jobs preserve failure isolation and +least-privilege permissions, while the substantive docs/build procedures are +delegated to repository commands rather than copied into YAML. + +Why adjacent scores do not fit: 3 does not fit because path and tool-bootstrap +knowledge is repeated on every routine trigger or tool-version update. 1 does +not fit because all copies remain confined to two small workflow files and the +substantive check authorities are still identifiable. + +Representative routine change: adding a new Rust-relevant `tools/**` tree would +require synchronized edits to +`.github/workflows/rust.yml:on.push.paths` and +`on.pull_request.paths`; updating the Rust checkout/toolchain baseline requires +reviewing the pins in every `rust.yml:jobs.*.steps` copy. + +## Lens-Boundary Notes + +- The repeated startup carriers in `fabro-workflow` could be labeled ownership + or simplicity. I counted their unclear amount of machinery under simplicity; + ownership was judged from whether each phase, resource lifecycle, and + dependency direction has a named home. +- The workflow's internal `Event` and durable `EventBody` have documented + distinct meanings. I therefore counted the many synchronized mappings under + duplication, not domain model. The separate `StageCompleted.status: String` + finding drives the domain-model score because it admits invalid states. +- Large web route files are not ownership findings merely because they are + large. They lower simplicity where common behavior is difficult to trace; the + ownership score moves only where several responsibilities remain concentrated + despite otherwise clear route/data/effect homes. +- In the web event layer, optional/untyped payload shape is a domain-model + finding. Repeating lifecycle-event policy across list, detail, and phase + consumers is a duplication finding. +- In CI, the stale `openapi/**` referent is a domain-model finding because the + path no longer means the API authority it purports to cover. Repeating trigger + and setup lists is separately a duplication finding. +- The `fabro-http` builder macro adds local indirection, but its primary effect + is to make shared async/blocking policy authoritative. I treated it as a + positive duplication mechanism rather than simplicity friction. diff --git a/.chisel/calibration/work/validation-1.md b/.chisel/calibration/work/validation-1.md new file mode 100644 index 000000000..6954086ae --- /dev/null +++ b/.chisel/calibration/work/validation-1.md @@ -0,0 +1,424 @@ +# Chisel calibration validation 1 + +Revision: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +This is an independent reading of only the requested assignments. Scores use the +mapped purposes and the final calibration rubric. Boundary evidence is included +where it establishes whether a scoped mechanism is on a production common path. + +## Summary + +| Component | Lens | Score | Evidence confidence | +|---|---|---:|---| +| `fabro-workflow` | `ownership-boundaries` | 2 | High | +| `fabro-workflow` | `domain-model` | 2 | High | +| `fabro-http` | `domain-model` | 4 | High | +| `fabro-http` | `duplication-knowledge` | 4 | High | +| `fabro-web-app` | `ownership-boundaries` | 4 | Medium | +| `repository-ci` | `ownership-boundaries` | 4 | Medium | +| `repository-ci` | `domain-model` | 2 | High | +| `fabro-checkpoint` | `ownership-boundaries` | 2 | High | +| `fabro-checkpoint` | `simplicity` | 4 | Medium | +| `fabro-checkpoint` | `domain-model` | 2 | High | +| `fabro-checkpoint` | `duplication-knowledge` | 3 | Medium | + +## `fabro-workflow` + +### `ownership-boundaries`: 2 + +- **Evidence:** `lifecycle/mod.rs:53-80` presents `WorkflowLifecycle` as the + callback owner, and `lifecycle/git.rs:77-93, 397-401` gives `GitLifecycle` + its own `last_git_sha` state. The normal `RunSession::run` path nevertheless + creates a second `last_git_sha`, reconstructs it by listening to emitted + checkpoint, terminal, and Git events, then passes it back into finalization + (`operations/start.rs:821-856, 914-923`). Terminal responsibility is split + again: engine outcomes become terminal events in + `pipeline/finalize.rs:524-596`, while bootstrap, initialization, and + finalization errors become `run.failed` through the outer operation in + `operations/start.rs:176-285, 288-346`. These crossings occur on the normal + run and error paths, not at an optional edge. +- **Strongest counterevidence:** `operations/start.rs:796-953` is a recognizable + top-level owner for the initialize → execute → finalize → pull-request + sequence, and `WorkflowLifecycle` explicitly orders focused delegates for + each executor callback (`lifecycle/mod.rs:221-469`). +- **Why adjacent scores do not fit:** 3 does not fit because the caller always + mirrors and resupplies Git identity on the common run path, and terminal + failure handling routinely selects between two owners. 1 does not fit because + both the executor callback owner and the outer run-session owner are stable + and traceable; the problem is their competition, not the absence of owners. +- **Rule discrimination:** Decision rule 2 is decisive for the mirrored + `last_git_sha`. The phrase “complete lifecycle” is otherwise ambiguous about + whether an executor lifecycle may end before durability finalization; the + explicit state round-trip makes the result 2 without relying on that + ambiguity. + +### `domain-model`: 2 + +- **Evidence:** The internal durable event shape stores + `Event::StageCompleted.status` as `String` + (`event/events.rs:264-272`). Both synthetic terminal-stage completion and + ordinary successful stage completion stringify the canonical + `StageOutcome` (`lifecycle/event.rs:215-240, 355-366`), after which the + mandatory event conversion reparses it and converts an unknown value to + `Failed` (`event/convert.rs:14-24, 309-333`). This typed → string → typed path + is part of every successful stage-completion event. +- **Strongest counterevidence:** `fabro_types::StageOutcome` is a stable + canonical type, most event fields are typed, and the fallback prevents an + unrecognized string from escaping into the stored projection. +- **Why adjacent scores do not fit:** 3 does not fit because common production + completion events depend on the invalid intermediate rather than using it as + a compatibility edge. 1 does not fit because the canonical status meaning is + clear and the conversion point is explicit. +- **Rule discrimination:** Decision rule 4 and the rubric's repository example + make this assignment unambiguous. + +## `fabro-http` + +### `domain-model`: 4 + +- **Evidence:** `ProxyPolicy` is a closed `System | Disabled` vocabulary; + parsing rejects every other boundary value + (`src/lib.rs:23-35`). Resolution gives explicit configuration precedence over + the environment, defaults absence to `System`, and rejects non-Unicode input + (`src/lib.rs:38-60`). Every async and blocking builder reaches that resolver + before construction (`src/lib.rs:160-166, 172-193`), while the deterministic + test helpers select the typed `Disabled` value + (`src/lib.rs:195-213`). +- **Strongest counterevidence:** The builder also exposes raw `no_proxy()` and + `proxy()` operations (`src/lib.rs:96-106`), so callers can combine an + underlying reqwest choice with `ProxyPolicy`; Unix-socket production callers + do use `no_proxy()` (`lib/foundation/fabro-client/src/client.rs:2123-2134`). +- **Why adjacent scores do not fit:** 3 does not fit because the common + policy-controlled constructors never interpret an invalid policy: they + return `HttpClientBuildError`. The raw builder operations represent valid + per-client transport configuration, not a second string vocabulary. 2 and 1 + do not fit because no common-path conversion or unstable meaning is present. +- **Rule discrimination:** Decision rule 4 is potentially non-discriminating + if every forwarded low-level builder method is called an “escape hatch.” + Here `no_proxy()` carries no invalid intermediate and does not weaken + `ProxyPolicy::resolve`, so treating it as ordinary typed builder + configuration preserves the rule's distinction. + +### `duplication-knowledge`: 4 + +- **Evidence:** `define_builder!` holds the complete shared async/blocking + builder policy once, including proxy resolution and construction + (`src/lib.rs:72-170`), and is instantiated for the two reqwest client kinds + (`src/lib.rs:172-193`). The four convenience constructors delegate to those + builders rather than reproducing policy (`src/lib.rs:195-213`). +- **Strongest counterevidence:** The generated facade necessarily lists each + forwarded reqwest method, and the test and non-test convenience constructors + have similar bodies. +- **Why adjacent scores do not fit:** 3 does not fit because the similar + forwarding and wrappers are syntax over one policy authority, not separately + maintained transport knowledge. 2 does not fit because a proxy-policy change + is made once in the macro/resolver, not synchronized across async and + blocking implementations. 1 does not fit because the authority is explicit. +- **Rule discrimination:** The rubric's `define_builder!` example directly + distinguishes shared macro expansion from semantic duplication; no material + ambiguity remains. + +## `fabro-web-app` + +### `ownership-boundaries`: 4 + +- **Evidence:** `entry.tsx:17-49` owns browser startup, chooses the normal or + installation route graph once, and installs shared SWR runtime policy. + `router.tsx:97-184` owns normal route composition. Shared transport and error + handling live in `lib/api-client.ts:64-160, 213-310`; shared reads such as + `useRun` and `useRunState` live in `lib/queries.ts:182-193`; run mutations and + their cache lifecycle live in `lib/mutations.ts:65-132`; and run-scoped SSE + subscription, invalidation, resync, and cleanup live in + `lib/run-events.ts:129-309`. The representative busy route composes those + owners rather than reimplementing them + (`routes/run-detail.tsx:79-145, 313-379`). +- **Strongest counterevidence:** Some route-local CRUD actions call the shared + API facade directly, and `run-detail.tsx:193-205` coordinates delete state, + cache invalidation, toast, and navigation in the route. +- **Why adjacent scores do not fit:** 3 does not fit because the counterevidence + is local page UX ownership; it does not split a shared transport, read, + mutation, or subscription lifecycle. 2 does not fit because routine run-page + changes use the established owners rather than coordinating competing ones. + 1 does not fit because startup, routing, transport, caching, and streaming + each have readily identifiable homes. +- **Rule discrimination:** “One owner” is mildly non-discriminating for a large + browser application unless responsibility is evaluated at lifecycle + granularity. Using the rubric's `apiData`/`useRun` example, route composition + is not itself a second owner. Confidence is Medium because this is the + largest sampled scope. + +## `repository-ci` + +### `ownership-boundaries`: 4 + +- **Evidence:** `rust.yml:3-40` owns Rust branch/PR/manual triggers and + concurrency, while its jobs contain format, lint, generated-doc, Linux test, + twin E2E, and manual macOS lifecycles (`rust.yml:48-147`). + `typescript.yml:3-34` owns the corresponding TypeScript triggers and + concurrency, and its jobs contain typecheck, test, and integrated SPA/Rust + build lifecycles (`typescript.yml:36-77`). Delegation to `cargo dev` is the + mapped dependency on build tooling, not reverse ownership. +- **Strongest counterevidence:** The TypeScript build invokes a Rust build + (`typescript.yml:75-77`), and invalid path selectors mean some intended + changes do not start the declared workflows. +- **Why adjacent scores do not fit:** 3 does not fit because the cross-language + build is the intentional embedded-SPA integration boundary, not friction, and + selector validity is classified under domain model by decision rule 6. 2 + does not fit because no routine job requires coordination between competing + CI owners. 1 does not fit because the two language validation homes and their + dependency direction are explicit. +- **Rule discrimination:** The score-4 phrase “complete lifecycle” is + non-discriminating for hosted CI if it is read to require repository + ownership of GitHub's runner lifecycle. This score treats the checked-in + trigger/job lifecycle as the mapped responsibility and the platform as an + intended boundary. + +### `domain-model`: 2 + +- **Evidence:** Both Rust trigger selectors name `openapi/**` + (`rust.yml:18,34`), but that revision has no tracked target there; the actual + API contract is `docs/public/api-reference/fabro-api.yaml`, which the + TypeScript client generation command consumes + (`lib/packages/fabro-api-client/package.json:7`). The real contract path is + absent from both workflow path filters. In addition, all three zizmor + `stale-action-refs` identifiers target `rust.yml:37`, `:49`, and `:62` + (`zizmor.yml:1-6`), which are respectively the end of trigger setup, the + `fmt` job key, and a `run` command—not action references at this revision. + These invalid identifiers sit directly in trigger and static-validation + configuration. +- **Strongest counterevidence:** The workflow/job vocabulary itself is stable, + all jobs and action pins have clear meanings, and changes under the large + valid Rust and TypeScript source selectors do trigger their expected suites. +- **Why adjacent scores do not fit:** 3 does not fit because the dead OpenAPI + selector is present in both routine branch and PR paths, while every scoped + zizmor exception lacks a current target. 1 does not fit because the overall + workflow and job model remains stable; the defect is a recurring set of + invalid identifiers. +- **Rule discrimination:** Decision rule 6 is decisive that these are domain + pressure rather than ownership or duplication. It does not state when one or + more dead selectors move from 3 to 2; centrality in both trigger modes and + total staleness of the scoped zizmor selectors supply that discrimination + here. + +## Control: `fabro-checkpoint` + +### `ownership-boundaries`: 2 + +- **Evidence:** The mapped component claims metadata branches, but its + production boundary consumer owns the metadata writer's branch, parent OID, + discovery, remote, and push lifecycle + (`fabro-workflow/src/run_metadata.rs:272-282, 313-439`). On every snapshot, + that caller validates entries, individually drives `Store` through blobs, + tree, commit, and ref update, and retains the parent identity for the next + write (`run_metadata.rs:313-350`). `BranchStore` provides a contained + read-modify-write owner (`branch.rs:17-24, 42-81`) but has no production + caller at this revision. +- **Strongest counterevidence:** The dependency direction is intended + (`fabro-workflow` depends on `fabro-checkpoint`), and the low-level `Store` + consistently owns Git object/ref operations (`git.rs:101-227`). +- **Why adjacent scores do not fit:** 3 does not fit because the lifecycle + crossing occurs on every metadata snapshot, not in an isolated adapter. 1 + does not fit because low-level Git ownership and the caller's higher-level + writer ownership are both stable; the problem is the split between them. +- **Rule discrimination:** Decision rule 2 applies because the caller retains + and resupplies branch/parent identity to complete successive writes. The + rubric does not say whether a deliberately low-level `Store` narrows the + mapped ownership claim; the explicit mapped claim to metadata branches makes + this crossing discriminating. + +### `simplicity`: 4 + +- **Evidence:** The production `Store` has direct blob, tree, commit, and ref + operations (`git.rs:123-226`). Tree conversion is a single read recursion and + a single bottom-up write path (`git.rs:229-310`). At the higher level, + `BranchStore::write_with` is a linear resolve → read → mutate → write → commit + → update sequence (`branch.rs:56-81`), and entry operations are small + delegates (`branch.rs:84-117`). Necessary Git layering is visible rather than + hidden behind competing configuration machinery. +- **Strongest counterevidence:** There are two entry levels, and the production + metadata writer uses the lower-level `Store` instead of `BranchStore`. +- **Why adjacent scores do not fit:** 3 does not fit because choosing the + low-level entry is required for replace-whole-tree and remote-parent behavior, + not unnecessary indirection. 2 does not fit because the scoped common + operations do not navigate competing implementations or configuration. 1 + does not fit because both paths are directly traceable. +- **Rule discrimination:** Ownership rule 2 could otherwise cause the + out-of-scope metadata writer's machinery to be counted again as simplicity + friction. The lens exclusions make that non-discriminating evidence here; + within the scoped implementation, the production primitives are direct. + +### `domain-model`: 2 + +- **Evidence:** `TreeEntries::set` accepts any `String` path without validation + (`git.rs:46-60`), and `write_tree` later interprets it by splitting on `/` + (`git.rs:149-153, 270-293`). The common metadata caller must therefore define + and apply `validate_metadata_path` outside this component before every + `TreeEntries` construction + (`fabro-workflow/src/run_metadata.rs:313-332, 471-480`). The component also + maps every unrecognized Git file mode to `Blob` + (`git.rs:21-35, 229-250`) rather than rejecting an unsupported state. +- **Strongest counterevidence:** `FileMode` is otherwise a closed enum, Git + object IDs use `git2::Oid`, and the current production metadata caller does + reject empty, absolute, dot-segment, and empty-segment paths before writing. +- **Why adjacent scores do not fit:** 3 does not fit because external path + validation is mandatory on every common metadata snapshot and the canonical + `TreeEntries` shape can always hold an invalid path. 1 does not fit because + the intended path and mode meanings remain clear and production does have a + validation step. +- **Rule discrimination:** Decision rule 4 clearly places the caller-validated + `TreeEntries` intermediate at 2. Whether unknown Git modes are a compatibility + escape hatch is ambiguous by itself, but it is not needed to choose the + score. + +### `duplication-knowledge`: 3 + +- **Evidence:** Branch-to-full-ref formatting is repeated in `Store::update_ref`, + `resolve_ref`, and `delete_ref` (`git.rs:182-225`), and the boundary metadata + writer has another `full_ref` transformation + (`fabro-workflow/src/run_metadata.rs:364-439`). `BranchStore::read_entry`, + `read_entries`, `list_entries`, and `tip_tree` also repeat parts of branch-tip + resolution (`branch.rs:119-184`). These repetitions are local and stable, but + there is no single helper enforcing them. +- **Strongest counterevidence:** Mutation sequencing is authoritative in + `BranchStore::write_with` (`branch.rs:56-81`), metadata branch naming has one + `META_BRANCH_PREFIX` constant (`lib.rs:7`), Git-author defaults have one + `Default` implementation (`author.rs:13-20`), and the repeated ref syntax is a + fixed Git protocol form rather than frequently changing Fabro policy. +- **Why adjacent scores do not fit:** 4 does not fit because ref normalization + and branch-tip traversal are still represented in several places. 2 does not + fit because there is no direct evidence that a routine checkpoint change + must alter those stable protocol transformations in sync; the repetitions are + isolated implementation knowledge. 1 does not fit because each policy has an + identifiable local authority even where a helper is absent. +- **Rule discrimination:** Decision rule 5 leaves a real 3-versus-4 ambiguity: + repeated `refs/heads/` can be classified as harmless protocol syntax. I score + 3 because the same branch-to-ref transformation crosses the component + boundary, but do not score 2 without evidence of routine synchronization. + +## Overall rubric observations + +- Decision rule 2 successfully distinguishes focused delegates from a lifecycle + that sends identity back through an event/caller round trip. +- Decision rule 6 prevents dead CI selectors from being double-counted as + ownership defects, but needs centrality/recurrence evidence to distinguish 2 + from 3. +- “One owner” and “complete lifecycle” need responsibility-sized interpretation + for route trees and hosted CI; otherwise healthy composition cannot reach 4. +- Decision rule 5 correctly keeps stable protocol repetition from automatically + becoming score 2, but the line between harmless syntax and a repeated + transformation remains the least discriminating part of this sample. + +## Round 2 revalidation + +| Component | Lens | Score | Confidence | +|---|---|---:|---| +| `fabro-http` | `duplication-knowledge` | 3 | Medium | +| `repository-ci` | `ownership-boundaries` | 2 | High | +| `fabro-checkpoint` | `ownership-boundaries` | 2 | High | +| `fabro-checkpoint` | `simplicity` | 3 | High | +| `fabro-checkpoint` | `domain-model` | 2 | High | +| `fabro-checkpoint` | `duplication-knowledge` | 3 | Medium | + +### `fabro-http` × `duplication-knowledge`: 3 + +- **Decisive evidence:** Proxy disabling has two concrete semantic + representations in the mapped entry layer: callers may set + `ProxyPolicy::Disabled` (`src/lib.rs:23-27, 90-94`), or call the separately + exposed `no_proxy()` builder operation (`src/lib.rs:96-100`). The former is + interpreted by calling the same underlying `inner.no_proxy()` transformation + during `build` (`src/lib.rs:160-165`). Both forms are used on direct boundary + paths: test constructors select the enum (`src/lib.rs:199-213`), while the + Unix-socket transport selects `no_proxy()` + (`lib/foundation/fabro-client/src/client.rs:2123-2134`). +- **Adjacent scores:** 4 does not fit revised rule 6 because there is a concrete + second representation of the same no-proxy decision. 2 does not fit because + an ordinary proxy-policy extension does not require manually synchronizing + those call sites; async and blocking policy construction still share the one + `define_builder!` mechanism (`src/lib.rs:72-193`). 1 does not fit because the + resolver remains a stable authority. +- **Remaining ambiguity:** `no_proxy()` can reasonably be viewed as a lower-level + reqwest operation rather than a second Fabro policy. Revised rule 6 makes 3 + the conservative result because `ProxyPolicy::Disabled` is implemented by + that exact operation, but this classification keeps confidence at Medium. + +### `repository-ci` × `ownership-boundaries`: 2 + +- **Decisive evidence:** The Rust check explicitly scans + `docs/public/api-reference/fabro-api.yaml` in its legacy-identity guard + (`rust.yml:80-92`), but neither push nor pull-request triggers include that + real path (`rust.yml:3-35`); they include the nonexistent `openapi/**` + selector instead (`rust.yml:18,34`). A routine API-contract change can + therefore change a scanned target without starting its owning check. +- **Adjacent scores:** 3 does not fit because the non-triggering target is on a + routine branch/PR check path, not an isolated manual edge. 1 does not fit + because the workflow, jobs, and intended trigger owner remain identifiable. + 4 is directly excluded by revised rule 3's trigger-coverage requirement. +- **Remaining ambiguity:** `typescript.yml:76` also invokes a Rust build from a + narrower trigger set, but that broader interpretation is unnecessary; the + explicitly scanned, non-triggering API contract is sufficient for 2. + +### `fabro-checkpoint` × `ownership-boundaries`: 2 + +- **Decisive evidence:** The mapped owner exposes low-level `Store` primitives, + while the routine metadata caller reconstructs the mapped branch lifecycle: + `RunMetadataWriter` owns branch, parent, and discovery state + (`fabro-workflow/src/run_metadata.rs:272-282`), then validates entries and + sequences blob, tree, commit, ref update, and retained parent state on every + snapshot (`run_metadata.rs:313-350`). No production boundary uses the + component's higher-level `BranchStore`. +- **Adjacent scores:** 3 does not fit because every metadata snapshot traverses + the split. 1 does not fit because the low-level Git owner and caller-side + lifecycle are both stable. 4 is directly excluded by revised rule 2: the + routine caller reconstructs a lifecycle the map assigns to this component. +- **Remaining ambiguity:** A narrower map that assigned only Git object + primitives to `fabro-checkpoint` could make this healthy delegation, but the + actual map explicitly assigns metadata branches and checkpoint commits. + +### `fabro-checkpoint` × `simplicity`: 3 + +- **Decisive evidence:** `Cargo.toml:16-24` carries `fabro-store` as a production + dependency, but scoped production code does not use it. The component also + exposes `BranchStore` as a parallel entry layer (`branch.rs:17-24`) that has + no production caller at this revision; the common metadata path uses `Store` + directly. The active `Store` path itself remains linear and direct + (`git.rs:123-226`). +- **Adjacent scores:** 4 is explicitly capped at 3 by revised rule 4 for the + unused production dependency and parallel unused entry layer. 2 does not fit + because routine production work does not repeatedly navigate those unused + elements; its `Store` path is direct. 1 does not fit because a stable common + path is easy to trace. +- **Remaining ambiguity:** Either isolated fact independently supplies the + revised rule's cap, so there is no material score ambiguity. + +### `fabro-checkpoint` × `domain-model`: 2 + +- **Decisive evidence:** `TreeEntries::set` accepts arbitrary string paths + (`git.rs:46-60`) before `write_tree` interprets them structurally + (`git.rs:149-153, 270-293`). Every common metadata snapshot must validate + those paths outside the mapped entry before constructing `TreeEntries` + (`fabro-workflow/src/run_metadata.rs:313-332, 471-480`). +- **Adjacent scores:** 3 does not fit revised rule 5 because caller validation + does not isolate an invalid-capable mapped entry used on every snapshot. 1 + does not fit because path meaning is stable and the caller does enforce it. + 4 is excluded because the canonical entry type itself admits invalid states. +- **Remaining ambiguity:** Unknown Git modes also collapse to `Blob` + (`git.rs:21-35`), but that compatibility question is not needed for the + score; the routine path shape is decisive. + +### `fabro-checkpoint` × `duplication-knowledge`: 3 + +- **Decisive evidence:** The short branch name is converted to + `refs/heads/{branch}` independently in `Store::update_ref`, `resolve_ref`, and + `delete_ref` (`git.rs:182-225`), while the routine boundary writer carries a + second `full_ref` conversion + (`fabro-workflow/src/run_metadata.rs:364-439`). These are concrete repeated + representations, but of stable Git protocol knowledge. +- **Adjacent scores:** 4 does not fit revised rule 6 because the + branch-to-full-ref transformation has a concrete second representation. 2 + does not fit because no ordinary mapped change is shown to require + synchronizing the stable Git namespace transformations; repeated call sites + alone are insufficient. 1 does not fit because the transformation and its + local authorities are clear. +- **Remaining ambiguity:** The literal can also be classified as harmless Git + syntax, which the lens excludes. Its repetition across the mapped boundary + supports 3, but the harmless-syntax distinction keeps confidence at Medium. diff --git a/.chisel/calibration/work/validation-2.md b/.chisel/calibration/work/validation-2.md new file mode 100644 index 000000000..b33efa10f --- /dev/null +++ b/.chisel/calibration/work/validation-2.md @@ -0,0 +1,450 @@ +# Chisel calibration validation 2 + +Revision reviewed: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c` + +This is an independent reading of the final rubric. I did not seek or infer +earlier scores. + +## Scores + +| Component | Lens | Score | Evidence confidence | +|---|---|---:|---| +| `fabro-workflow` | `ownership-boundaries` | 2 | High | +| `fabro-workflow` | `domain-model` | 2 | High | +| `fabro-http` | `domain-model` | 4 | High | +| `fabro-http` | `duplication-knowledge` | 4 | Medium | +| `fabro-web-app` | `ownership-boundaries` | 4 | Medium | +| `repository-ci` | `ownership-boundaries` | 2 | High | +| `repository-ci` | `domain-model` | 2 | High | +| `fabro-checkpoint` | `ownership-boundaries` | 4 | Medium | +| `fabro-checkpoint` | `simplicity` | 4 | Medium | +| `fabro-checkpoint` | `domain-model` | 3 | Medium | +| `fabro-checkpoint` | `duplication-knowledge` | 2 | Medium | + +## Disputed assignments + +### `fabro-workflow` × `ownership-boundaries` — 2 + +**Direct evidence.** `WorkflowLifecycle` is a real central owner for engine +callback ordering: it contains the event, hook, fidelity, status, circuit +breaker, git, and artifact delegates and orders them in every callback +(`src/lifecycle/mod.rs:53-80`, `223-470`). The full run lifecycle nevertheless +crosses that owner on normal paths. `WorkflowLifecycle::on_run_end` only runs +the hook (`src/lifecycle/mod.rs:467-469`); `pipeline::finalize` separately builds +and emits the terminal event and stops the sandbox +(`src/pipeline/finalize.rs:524-635`); `RunSession::run` separately owns +initialize/execute/finalize, progress flushing, steering drain, and a second +sandbox cleanup guard (`src/operations/start.rs:796-953`); detached bootstrap +and completion guards own additional terminal-failure paths +(`src/operations/start.rs:956-1139`). A routine change to terminal ordering or +cleanup must account for these owners. + +**Strongest counterevidence.** The split is deliberate. In particular, +`finalize` documents why the terminal event must follow metadata flushing, and +the scope guards cover panic/interruption paths that an async lifecycle callback +cannot reliably cover. + +**Why adjacent scores do not fit.** Score 3 does not fit because the split is on +every ordinary terminal path, not an isolated compatibility path. Score 1 does +not fit because the owners and dependency direction are identifiable: +`RunSession` is the outer orchestrator and `WorkflowLifecycle` consistently owns +engine callbacks. + +**Rule discrimination.** Decision rule 2 is useful here, but “complete routine +lifecycle operations” must include terminal emission and resource cleanup, not +only engine callbacks. Without that reading, the positive orchestrator example +could make 3 and 2 hard to distinguish. + +### `fabro-workflow` × `domain-model` — 2 + +**Direct evidence.** The canonical execution result is the typed +`StageOutcome`, re-exported in `src/outcome.rs:1-12`. The common stage-completion +event instead stores `status: String` (`src/event/events.rs:264-293`). +`EventLifecycle::after_node` converts the typed value to a string for every +successful completion (`src/lifecycle/event.rs:319-378`), and +`event_body_from_event` reparses it into `StageOutcome` +(`src/event/convert.rs:309-348`). Unknown strings are silently reinterpreted as +a non-retryable failure (`src/event/convert.rs:14-24`). The same string +intermediate is used for synthetic terminal stages +(`src/lifecycle/event.rs:183-242`). + +**Strongest counterevidence.** Durable `fabro_types::StageCompletedProps` is +typed, and ordinary producers derive the string from a typed value rather than +accepting arbitrary user text. + +**Why adjacent scores do not fit.** Score 3 does not fit because the conversion +and invalid intermediate occur on the common event path for every completed +stage. Score 1 does not fit because `StageOutcome` supplies a stable canonical +meaning and most execution code uses it directly. + +**Rule discrimination.** Decision rule 4 and the repository example are +decisive. The rule would be non-discriminating if “compatibility escape hatch” +were allowed to describe the central `Event` type merely because the durable +type is healthier. + +### `fabro-http` × `domain-model` — 4 + +**Direct evidence.** `ProxyPolicy` is a closed two-variant vocabulary +(`src/lib.rs:23-27`). The environment boundary parses case-insensitively and +rejects every other value with a typed `HttpClientBuildError` +(`src/lib.rs:29-70`). Explicit policy has a documented precedence in +`resolve_with_env_value`, and both async and blocking builders resolve the +policy immediately before applying it (`src/lib.rs:38-59`, `160-166`, +`172-193`). The common production and test constructors all pass through those +builders (`src/lib.rs:195-213`). + +**Strongest counterevidence.** The builders also expose the lower-level +`no_proxy()` and `proxy()` methods (`src/lib.rs:96-106`), so callers can express +transport configuration outside the high-level enum. + +**Why adjacent scores do not fit.** Score 3 does not fit because the lower-level +methods are intentional reqwest-facade escape hatches; the common constructors +and environment boundary do not rely on an invalid or ambiguous policy value. +There is positive production enforcement rather than a test-only contract. + +**Rule discrimination.** Decision rule 4 discriminates well if “low-level +escape hatch” is read literally. If any alternate builder method were treated +as a second domain meaning, scores 3 and 4 would become difficult to distinguish +for facades. + +### `fabro-http` × `duplication-knowledge` — 4 + +**Direct evidence.** `define_builder!` is one production mechanism for all +shared async/blocking builder methods and for applying proxy policy +(`src/lib.rs:72-170`); the two concrete builders are declarations of that +mechanism (`src/lib.rs:172-193`). `ProxyPolicy::resolve` is the single authority +for explicit-versus-environment precedence (`src/lib.rs:38-59`), and the four +convenience constructors delegate to the builders (`src/lib.rs:195-213`). +Workspace boundary evidence reinforces this authority: `clippy.toml` disallows +raw reqwest client constructors in favor of these functions/builders. + +**Strongest counterevidence.** The tokens `system` and `disabled` also appear in +the human-readable error text, and the async/blocking test constructors repeat +the choice of `ProxyPolicy::Disabled`. + +**Why adjacent scores do not fit.** Score 3 does not fit because the repeated +tokens and two one-line convenience constructors do not form independent +authorities for a recurring transformation. The macro and resolver are what +enforce behavior. + +**Rule discrimination.** Decision rule 5 is useful but leaves a small judgment +gap around repeated diagnostic vocabulary. Here that repetition is +non-discriminating: adding a variant would make the exhaustive application +match fail to compile, while one diagnostic sentence is not a second policy +engine. This is why confidence is Medium rather than High. + +### `fabro-web-app` × `ownership-boundaries` — 4 + +**Direct evidence.** Shared HTTP configuration, authentication redirect, and +error normalization live in `app/lib/api-client.ts:64-160,213-309`. Read state +and cache keys live in `app/lib/queries.ts` and +`app/lib/query-keys.ts`; for example, `useRun` owns the run-detail fetch/cache +lifecycle (`queries.ts:182-187`). Shared run mutations and their cache updates +live in `app/lib/mutations.ts:42-208`. Run SSE connection sharing, cleanup, and +cache invalidation live in `app/lib/sse.ts:42-189` and +`app/lib/run-events.ts:129-308`. Browser resources with more specialized +lifecycles are likewise contained: terminal WebSocket/xterm/listener cleanup is +in `app/hooks/use-terminal-session.ts:62-229`, and install polling owns its +timer, interval, and abort controller in +`app/hooks/use-install-effects.ts:72-127`. + +`RunDetail` composes these owners and retains view-local state and interaction +ordering (`app/routes/run-detail.tsx:79-145,148-379`). Its size does not make it +the owner of transport or resource cleanup. + +**Strongest counterevidence.** Several feature routes perform feature-local +create/edit/delete calls and SWR invalidation directly, and `RunDetail` owns the +delete dialog, pending state, toast, list invalidation, and navigation +(`run-detail.tsx:193-205`) rather than using a single mutation hook for that +entire interaction. + +**Why adjacent scores do not fit.** Score 3 does not fit without a concrete +isolated lifecycle that has competing owners. The direct route mutations keep +their feature interaction lifecycle local and still use the shared transport; +they are not evidence that ordinary reads, SSE, or browser resources leak into +route composition. + +**Rule discrimination.** The final repository example is discriminating: +“busy route” must not itself count as boundary leakage. Confidence remains +Medium because the application scope is broad, although the representative +read, mutation, live-update, terminal, install, and route boundaries converge. + +### `repository-ci` × `ownership-boundaries` — 2 + +**Direct evidence.** The Rust workflow’s Clippy job owns a repository-wide +“legacy auth identity removal” guard that scans `lib/apps`, `lib/components`, +`lib/foundation`, `apps`, `lib/packages`, and the OpenAPI document +(`.github/workflows/rust.yml:80-91`). The workflow’s path filters do not include +`apps/**`, `lib/packages/**`, or +`docs/public/api-reference/fabro-api.yaml` +(`rust.yml:3-35`). A routine change in a scanned TypeScript/package/API path can +therefore introduce a forbidden identity without starting the job that owns the +guard. The policy lifecycle is placed under a narrower Rust trigger than the +responsibility it claims. + +**Strongest counterevidence.** The primary Rust and TypeScript build/test +responsibilities otherwise have clear workflow homes, read-only permissions, +and stable concurrency ownership (`rust.yml:38-147`; +`typescript.yml:30-77`). The TypeScript production build’s Rust step is a +legitimate composition point because it builds the Rust binary with the +embedded SPA. + +**Why adjacent scores do not fit.** Score 3 does not fit because the trigger +mismatch affects ordinary changes in multiple scanned source areas, not an +isolated maintenance path. Score 1 does not fit because the two main language +workflows and their jobs still have stable owners and dependency direction. + +**Rule discrimination.** No final rule explicitly says how to classify a check +whose declared scan scope exceeds its trigger scope. The ownership lens’s +“complete lifecycle” language is sufficient, but an explicit trigger/target +coverage rule would make 2 versus 3 less ambiguous. + +### `repository-ci` × `domain-model` — 2 + +**Direct evidence.** Every value in `.github/zizmor.yml` is a line-addressed +identifier: `rust.yml:37`, `rust.yml:49`, and `rust.yml:62` +(`.github/zizmor.yml:1-6`). At this revision those lines are respectively a +blank separator, the `fmt` job key, and a `run:` step—not action references. +Thus none is a current target for the configured `stale-action-refs` ignores. +Routine edits to `rust.yml` can change the accidental referents again without +changing the selectors. + +**Strongest counterevidence.** The syntax still communicates an intended +workflow-and-line selector, and the main workflow job/status vocabulary is +otherwise stable. + +**Why adjacent scores do not fit.** Score 3 does not fit because all three +values in the entire scoped zizmor configuration lack their intended current +referent; this is not one isolated compatibility value. Score 1 does not fit +because the selector format and intended concept remain identifiable even +though the instances are stale. + +**Rule discrimination.** Decision rule 6 is decisive and correctly keeps this +under domain model rather than ownership. It would not by itself distinguish 2 +from 3; the fact that every configured identifier is stale and line edits make +the condition recur supplies that distinction. + +## Control: `fabro-checkpoint` + +### `fabro-checkpoint` × `ownership-boundaries` — 4 + +**Direct evidence.** `git::Store` owns the `git2::Repository` and the low-level +blob/tree/commit/ref operations (`src/git.rs:101-227`). +`branch::BranchStore` owns branch identity, author identity, and the complete +local read-modify-write lifecycle, including parent resolution, tree read, +commit, and ref update (`src/branch.rs:17-82`). Author and trailer concerns are +focused modules rather than state hidden in callers (`src/author.rs`; +`src/trailer.rs`). Boundary evidence points in the intended direction: +`fabro-workflow` depends on these primitives, while its +`RunMetadataWriter` owns the additional temp repository, remote discovery, +credentials, push, and degradation lifecycle. That is a higher-level owner +using a lower-level delegate, not a reverse dependency. + +**Strongest counterevidence.** The production metadata writer uses `Store` +directly and manually sequences blob, tree, commit, and ref operations +(`fabro-workflow/src/run_metadata.rs:313-361`) instead of using `BranchStore`. +The crate name/description can make that look like the mapped checkpoint +lifecycle has escaped the component. + +**Why adjacent scores do not fit.** Score 3 does not fit if responsibilities are +classified by their actual state: `Store` owns local Git mechanics, +`BranchStore` owns local branch writes, and `RunMetadataWriter` owns remote run +metadata. No concrete resource is acquired by one of those owners and released +by another. + +**Rule discrimination.** Decision rule 2 is ambiguous for intentionally +low-level facades. Passing a branch to `Store::update_ref` should not alone mean +“resupplying identity” when the caller owns the higher-level remote branch +lifecycle and `Store` never claimed it. If the mapped purpose is instead read +as all run-checkpoint lifecycle, this assignment could become 2; that purpose +boundary should be fixed before using the control for strict agreement. + +### `fabro-checkpoint` × `simplicity` — 4 + +**Direct evidence.** The local branch write path is linear in +`BranchStore::write_with`: resolve parent, read tree, apply one caller mutation, +write tree, commit, update ref (`src/branch.rs:56-81`). Single-file, +multi-file, and delete operations are thin delegates to that path +(`src/branch.rs:84-117`). The lower-level tree conversion is one direct +flat-to-nested algorithm (`src/git.rs:229-309`), and trailer formatting/parsing +uses straightforward local control flow (`src/trailer.rs:9-87`). + +**Strongest counterevidence.** `BranchStore` has no external production caller +at this revision; the actual metadata path uses the lower-level `Store` API. +There is also some unused-looking surface such as `MetadataError` and generic +branch read/list/log helpers. + +**Why adjacent scores do not fit.** Score 3 does not fit because no direct +production evidence shows routine changes navigating the unused surface or +competing implementations. The production `Store` call sequence is itself +linear. The rubric explicitly says a public method alone does not establish +frequency, so unused API breadth cannot by itself create common-path +indirection. + +**Rule discrimination.** The score-4 requirement for a “production mechanism” +is mildly ambiguous when the clearest high-level mechanism has no production +caller but its lower-level mechanism does. Treating compiled non-test code as +sufficient would make the rule non-discriminating; this score instead relies on +the directly used `Store` path also being traceable. + +### `fabro-checkpoint` × `domain-model` — 3 + +**Direct evidence.** The common metadata boundary validates every path before +putting it into `TreeEntries` +(`fabro-workflow/src/run_metadata.rs:319-336,471-481`), explicitly selects +`FileMode::Blob`, and converts author strings with the fallible +`git2::Signature::now` before committing (`run_metadata.rs:337-345`). Within the +control, `FileMode` and `TreeEntries` give Git tree entries a stable meaning +(`src/git.rs:13-99`), and Git failures stay typed (`src/error.rs:3-32`). + +There is nevertheless isolated model friction. `TreeEntries::set` accepts any +string path with no invariant-bearing path type (`src/git.rs:59-61`); +`FileMode::from_i32` maps every unrecognized Git mode to `Blob` +(`src/git.rs:30-35`); `GitAuthor` has public raw string fields +(`src/author.rs:6-11`); and `BranchStore` says trees grow monotonically while +also exposing `delete_entry` (`src/branch.rs:17-19,111-117`). + +**Strongest counterevidence.** These are not merely hypothetical invalid +shapes: low-level public callers can bypass the production metadata-path +validation, and Git supports meaningful modes omitted by `FileMode`. + +**Why adjacent scores do not fit.** Score 4 does not fit because the low-level +types themselves do not reject invalid paths/authors or preserve every Git +mode. Score 2 does not fit because the directly traced production metadata path +validates before interpretation and does not depend on the fallback +`from_i32`; the friction is in lower-level escape paths and the currently +unused `BranchStore`, not every common snapshot. + +**Rule discrimination.** Decision rule 4 is useful but ambiguous about whether +a common caller validating raw values before a low-level API counts as a +“common-path invalid intermediate.” The rule should distinguish an actually +reparsed/ambiguous value from a raw value that has already passed one boundary +check but lacks an invariant-bearing Rust type. + +### `fabro-checkpoint` × `duplication-knowledge` — 2 + +**Direct evidence.** The branch-name-to-full-ref transformation +`refs/heads/{branch}` is repeated independently in `Store::update_ref`, +`Store::resolve_ref`, and `Store::delete_ref` +(`src/git.rs:182-225`). The direct production boundary repeats it again in +`RunMetadataWriter::full_ref` +(`fabro-workflow/src/run_metadata.rs:425-439`). A routine addition or change to +branch ref handling must preserve the same transformation in each location. +The trailer grammar has a second, smaller recurrence: `": "` is independently +formatted, parsed, and detected in `append`, `parse`, `format_message`, and +`has_trailing_trailer_block` (`src/trailer.rs:11-12,28-40,45-59,68-86`). + +**Strongest counterevidence.** Both grammars are tiny and stable, tests cover +the trailer forms, and the three Store methods currently agree. A helper could +look like cosmetic deduplication rather than a material abstraction. + +**Why adjacent scores do not fit.** Score 3 does not fit because branch +resolution/update/deletion are ordinary Store operations and direct boundary +code already supplies a fourth recurrence; this is not only a hypothetical +future variant. Score 1 does not fit because the repeated transformations are +stable and readily identifiable even though they lack a single authority. + +**Rule discrimination.** Decision rule 5 is decisive only if “direct evidence +of routine recurrence” includes several current operations applying the same +transformation. If it instead requires historical change evidence, the final +rule would be non-discriminating for a revision-only review and this assignment +would move toward 3. + +## Round 2 revalidation + +These scores supersede the corresponding Round 1 scores. + +### `fabro-http` × `duplication-knowledge` — 3 (Medium) + +**Decisive evidence.** `ProxyPolicy::parse` is the behavioral authority for the +external `system`/`disabled` vocabulary, while +`HttpClientBuildError::InvalidProxyPolicy` separately enumerates those values +in its diagnostic (`src/lib.rs:29-35,63-66`). The builder macro remains one +authority for applying the policy to both client kinds (`src/lib.rs:72-193`). + +**Adjacent scores and ambiguity.** Score 4 does not fit because the diagnostic +is a concrete second representation that can drift. Score 2 does not fit +because proxy behavior is not independently reimplemented: the shared +resolver and macro enforce it, and the two no-proxy convenience constructors +are call sites rather than separate authorities (`src/lib.rs:195-213`). The +remaining ambiguity is whether changing the closed proxy vocabulary is routine +enough to make the diagnostic synchronization central; I treat it as isolated. + +### `repository-ci` × `ownership-boundaries` — 2 (High) + +**Decisive evidence.** The Rust workflow's legacy-auth check scans `apps`, +`lib/packages`, and `docs/public/api-reference/fabro-api.yaml` +(`rust.yml:80-91`), but its push and pull-request path filters omit all three +(`rust.yml:3-35`). Under decision rule 3, that check owns trigger coverage for +every path it scans, so routine changes in those targets bypass its lifecycle. + +**Adjacent scores and ambiguity.** Score 3 does not fit because the missing +triggers affect several routine source and contract paths, not an isolated +edge. Score 1 does not fit because the Rust and TypeScript workflow owners and +dependency direction remain stable. No material ambiguity remains under the +new trigger-coverage rule. + +### `fabro-checkpoint` × `ownership-boundaries` — 2 (High) + +**Decisive evidence.** The map assigns checkpoint commits, trees, metadata +branches, authorship, and trailers to this component. The routine +`RunMetadataWriter` caller reconstructs that mapped lifecycle from `Store` +primitives: it writes blobs and a tree, creates the commit and author/message, +updates the ref, and pushes +(`fabro-workflow/src/run_metadata.rs:313-361`). Decision rule 2 therefore +places ownership at 2 even though the crate dependency points toward +`fabro-checkpoint`. + +**Adjacent scores and ambiguity.** Score 3 does not fit because this is the +common metadata snapshot path, not an edge case. Score 1 does not fit because +the dependency direction and the low-level `Store` role are stable, and +`BranchStore::write_with` demonstrates a coherent lifecycle owner inside the +crate (`src/branch.rs:56-81`). The only remaining ambiguity is how specialized +the metadata commit is, but the map explicitly includes metadata branches. + +### `fabro-checkpoint` × `simplicity` — 3 (High) + +**Decisive evidence.** `fabro-store` and `serde` are production dependencies +with no source use (`Cargo.toml:16-24`), and `BranchStore` is a parallel +high-level entry layer with no production caller outside this crate. Decision +rule 4 makes those isolated simplicity frictions and caps 4 at 3. + +**Adjacent scores and ambiguity.** Score 4 does not fit because the unused +production edges and parallel layer are concrete. Score 2 does not fit because +the production `Store` path remains direct; normal callers do not navigate the +unused dependencies or `BranchStore`. Whether `BranchStore` is retained for a +future caller is ambiguous, but the unused dependencies alone sustain 3. + +### `fabro-checkpoint` × `domain-model` — 2 (Medium) + +**Decisive evidence.** The mapped Git-tree entry accepts any `String` path +through `TreeEntries::set` (`src/git.rs:44-61`), while the routine metadata +writer must validate paths before constructing those entries +(`fabro-workflow/src/run_metadata.rs:319-336,471-481`). Decision rule 5 says +caller validation does not isolate an invalid-capable mapped entry. +`FileMode::from_i32` also collapses every unrecognized mode to `Blob` +(`src/git.rs:29-35`). + +**Adjacent scores and ambiguity.** Score 3 does not fit because raw paths cross +the common write boundary. Score 1 does not fit because tree entries, modes, +and authors retain stable meanings and the caller does validate its input. +Confidence is Medium because a deliberately low-level Git store can reasonably +leave some path constraints to higher-level schemas, although the revised rule +weighs against that interpretation. + +### `fabro-checkpoint` × `duplication-knowledge` — 3 (Medium) + +**Decisive evidence.** The `refs/heads/{branch}` transformation is repeated in +three `Store` operations and once at the workflow boundary +(`src/git.rs:182-225`; `fabro-workflow/src/run_metadata.rs:425-439`). +Trailer formatting, parsing, and block detection also encode the `": "` +convention separately (`src/trailer.rs:11-12,28-40,45-59,68-86`). These are +concrete second representations, so decision rule 6 caps 4 at 3. + +**Adjacent scores and ambiguity.** Score 2 does not fit on the current evidence: +adding a Store operation or trailer key may repeat a call-site convention, but +does not require an ordinary mapped change to modify all existing locations. +Score 4 does not fit because the representations are nevertheless concrete and +can drift. The remaining ambiguity is whether broader trailer-syntax support +would be routine maintenance; if so, its formatter/parser/detector +synchronization would support 2. diff --git a/.chisel/calibration/work/validation-3.md b/.chisel/calibration/work/validation-3.md new file mode 100644 index 000000000..3969b0b13 --- /dev/null +++ b/.chisel/calibration/work/validation-3.md @@ -0,0 +1,447 @@ +# Chisel calibration validation 3 + +Revision reviewed: `6bb6b5efcc0e36b52e3c097f532d9f2c00914c6c`. + +This is an independent reading of the final rubric and the assigned component +scopes. I traced representative production entry points and direct boundary +callers. I did not inspect prior calibration scores or any other file in +`.chisel/calibration/work/`. + +## Score summary + +| Component | Lens | Score | Evidence confidence | +| --- | --- | ---: | --- | +| `fabro-workflow` | ownership-boundaries | 2 | High | +| `fabro-workflow` | domain-model | 2 | High | +| `fabro-http` | domain-model | 4 | High | +| `fabro-http` | duplication-knowledge | 3 | Medium | +| `fabro-web-app` | ownership-boundaries | 4 | Medium | +| `repository-ci` | ownership-boundaries | 4 | Medium | +| `repository-ci` | domain-model | 2 | High | +| `fabro-checkpoint` | ownership-boundaries | 2 | High | +| `fabro-checkpoint` | simplicity | 3 | Medium | +| `fabro-checkpoint` | domain-model | 2 | High | +| `fabro-checkpoint` | duplication-knowledge | 3 | Medium | + +## `fabro-workflow` + +### `ownership-boundaries`: 2 (High) + +- **Evidence:** `src/lifecycle/mod.rs:53-80,221-469` provides a real central + `WorkflowLifecycle` and explicitly orders focused event, hook, fidelity, Git, + artifact, status, and circuit-breaker delegates. Its terminal callback, + however, only forwards `on_run_end` to the hook. Normal terminal persistence, + metadata completion, terminal event emission, and sandbox stopping instead + live in `src/pipeline/finalize.rs:524-635`. Bootstrap and execution failures + take another terminal path in `src/operations/start.rs:176-345`, while + `RunSession::run` also installs cleanup and drain guards at + `src/operations/start.rs:889-947`. A routine terminal-lifecycle change must + therefore coordinate the lifecycle orchestrator, finalizer, and detached + failure/guard paths. +- **Strongest counterevidence:** The normal phase sequence is plainly owned by + `RunSession::run` (`initialize -> execute -> finalize -> pull_request`), and + callback ordering inside graph execution has one obvious owner, + `WorkflowLifecycle`. +- **Why 3 does not fit:** Terminal completion, failure, persistence, and cleanup + are common paths, not isolated edge compatibility. The split therefore + remains central even though each individual phase is understandable. +- **Why 1 does not fit:** Stable phase owners and a stable dependency direction + are readily identifiable; the problem is coordination among them, not the + absence of ownership. +- **Rule discrimination:** The repository example correctly requires terminal + inspection and rule 1 makes the common terminal split score-capping. Decision + rule 2 is less literal here because no single identity is resupplied across + every split, but the score does not depend on that rule. + +### `domain-model`: 2 (High) + +- **Evidence:** `src/lifecycle/event.rs:319-390` starts with the typed + `StageOutcome` on an `Outcome`, serializes it with + `outcome.status.to_string()`, and stores the result in the + `Event::StageCompleted.status: String` field declared at + `src/event/events.rs:264-293`. Every successful stage then passes through + `src/event/convert.rs:14-24,309-348`, which reparses the string and silently + converts an unknown value into a non-retryable failure. This is the ordinary + durable-event path, not an import-only compatibility path. +- **Strongest counterevidence:** The destination event model already has the + canonical `fabro_types::StageOutcome`, parallel-branch completion carries it + directly, and other core run concepts use typed IDs, reasons, timings, and an + opaque `ResumeState` (`src/pipeline/types.rs:252-285`). +- **Why 3 does not fit:** The invalid intermediate occurs for each ordinary + successful stage before durable interpretation, so it is central rather than + an isolated escape hatch. +- **Why 1 does not fit:** `StageOutcome` itself has a stable, typed meaning; the + defect is the recurring string round trip between two typed points. +- **Rule discrimination:** Decision rule 4 is directly discriminating here: + this is exactly a common-path invalid intermediate. + +## `fabro-http` + +### `domain-model`: 4 (High) + +- **Evidence:** `src/lib.rs:23-61` gives proxy behavior a closed + `ProxyPolicy::{System, Disabled}` vocabulary. The environment boundary + accepts case-insensitive valid names, rejects every other value with a typed + `HttpClientBuildError`, handles non-Unicode values explicitly, gives explicit + policy precedence over the environment, and resolves absence to `System`. + Both generated builders invoke this resolver before constructing a client + (`src/lib.rs:72-193`), and the test-client entry points select + `ProxyPolicy::Disabled` rather than passing an unchecked string + (`src/lib.rs:195-213`). +- **Strongest counterevidence:** The facade deliberately exposes reqwest's + lower-level `Proxy` and `.no_proxy()` operations, so callers can compose + transport details outside the two-value environment policy. +- **Why 3 does not fit:** Those operations are typed builder choices, not + unvalidated representations of the `FABRO_HTTP_PROXY_POLICY` value. Every + common construction path still validates that boundary before use; I found no + material meaning or validation friction. +- **Why 1-2 do not fit:** There is one stable meaning, one resolver, and no + recurring conversion through an invalid intermediate. +- **Rule discrimination:** Decision rule 4 could be read ambiguously if every + low-level builder method is called a policy escape hatch. The rubric's own + `ProxyPolicy` example resolves that ambiguity in favor of the closed, + validated environment-policy model. + +### `duplication-knowledge`: 3 (Medium) + +- **Evidence:** `define_builder!` at `src/lib.rs:72-193` is one authoritative + production mechanism for the shared async/blocking builder surface and for + applying the resolved proxy policy. The four convenience constructors route + through those builders. The remaining repeated knowledge is narrow: + `"system"` and `"disabled"` appear both in the parser and in the manually + maintained `InvalidProxyPolicy` expectation text + (`src/lib.rs:29-35,63-69`). +- **Strongest counterevidence:** The macro removes the materially risky + async/blocking synchronization, and the compiler forces the policy-application + match to cover every enum variant. The two test helpers' use of + `ProxyPolicy::Disabled` is ordinary reuse, not a second policy authority. +- **Why 4 does not fit:** The user-facing valid-value list is a small second + representation that can drift from the parser, so there is some isolated + repeated domain knowledge. +- **Why 2 does not fit:** There is no direct evidence that routine changes + repeatedly synchronize separate async/blocking implementations. A future + enum variant is hypothetical, and rule 5 specifically says exhaustive + compiler-checked branches and hypothetical variants do not establish + competing authorities. +- **Rule discrimination:** Rule 5 cleanly rules out 2 but is non-discriminating + between 3 and 4 for a duplicated allowed-value error message. I treat that + message as real but isolated maintenance friction, hence 3. + +## `fabro-web-app` + +### `ownership-boundaries`: 4 (Medium) + +- **Evidence:** `app/entry.tsx:17-48` owns root creation, global SWR policy, + build-version guarding, toast mounting, and the single normal/install router + choice. `app/router.tsx:97-184` owns normal route composition, while + `app/install-router.tsx:6-22` owns the install graph. Shared HTTP translation + and unauthorized handling live in `app/lib/api-client.ts:213-309`; shared + reads such as `useRun` live in `app/lib/queries.ts:182-187`; recurring run + mutations and cache follow-up live in + `app/lib/mutations.ts:65-149`. Route components compose these owners. + Separately, `scripts/build.ts:183-249,289-368` contains the complete + app-local build, atomic publication, and old-build pruning lifecycle and + publishes only `apps/fabro-web/dist`; boundary tooling mirrors that output + into the Rust SPA rather than the web build writing across the boundary. +- **Strongest counterevidence:** Some route-specific CRUD mutations import + `apiData` and generated API objects directly, and the install feature spans + `install-app.tsx`, `install-api.ts`, `install-query.ts`, and effect hooks. + `run-detail.tsx` is also a busy composition point. +- **Why 3 does not fit:** The direct calls remain at the route-specific UX + owner and still use the shared transport/error boundary; shared read and + recurring run-lifecycle responsibilities are not reimplemented there. + Install state, transport, query, and browser effects have distinct homes. + I found no isolated lifecycle that must leave its owner and resupply identity. +- **Why 1-2 do not fit:** Runtime, routing, transport, queries, route UX, and + build publication all have stable owners with dependencies pointing from + composition toward shared services. +- **Rule discrimination:** The final repository example is useful and + discriminating: a large route is not by itself boundary leakage. The score + would change if direct routes reimplemented shared transport or cache + lifecycles, but representative boundary checks did not show that. + +## `repository-ci` + +### `ownership-boundaries`: 4 (Medium) + +- **Evidence:** `.github/workflows/rust.yml:48-147` owns Rust formatting, + lint/architecture checks, generated docs, Linux tests, twin E2E selection, + and manual macOS tests. `.github/workflows/typescript.yml:36-77` owns web and + generated-client typechecks, web tests, and the production embedded-SPA + integration build. Each workflow owns its concurrency and least-privilege job + permissions. The TypeScript workflow's `cargo dev build` is the intentional + integration boundary that consumes the web bundle; it does not create a + competing implementation of the web build. +- **Strongest counterevidence:** The TypeScript build job invokes Rust build + tooling, path scopes overlap around `lib/apps/fabro-spa/**`, and + `.github/zizmor.yml` is configuration whose consumer is not shown in these + files. +- **Why 3 does not fit:** Cross-language integration is part of the mapped CI + purpose and has one concrete home. The stale configuration values discussed + below are domain-model findings, while duplicated push/pull selectors are + duplication findings; counting either again as ownership friction would + violate the rubric's primary-lens rule. +- **Why 1-2 do not fit:** The Rust and TypeScript responsibilities and their + dependency direction are stable. Routine validation changes have an obvious + workflow owner rather than requiring competing lifecycle owners. +- **Rule discrimination:** The instruction not to penalize an unevidenced + missing lifecycle matters for the unseen zizmor consumer. The rubric is + otherwise discriminating once repeated selector policy is kept out of the + ownership lens. + +### `domain-model`: 2 (High) + +- **Evidence:** Both Rust trigger selectors name `openapi/**` + (`.github/workflows/rust.yml:6-19,22-35`), but that revision has no tracked + `openapi/` target. The actual Rust generator and TypeScript generator consume + `docs/public/api-reference/fabro-api.yaml` + (`lib/foundation/fabro-api/build.rs:159` and + `lib/packages/fabro-api-client/package.json:7`), a path omitted from both + workflow trigger models. This makes a core API-spec change invisible to the + intended CI trigger. In addition, all three + `.github/zizmor.yml:4-6` line selectors target + `.github/workflows/rust.yml` lines 37, 49, and 62, which are respectively + `workflow_dispatch`, the `fmt` job key, and a shell `run`, not action + references for `stale-action-refs`. +- **Strongest counterevidence:** Most configured branches, paths, action SHAs, + runner labels, job names, and commands have clear current targets, and both + workflow documents have a stable overall schema. +- **Why 3 does not fit:** The dead OpenAPI selector sits in both central Rust + push and pull-request triggers and omits the actual source of truth. It is not + merely an isolated stale lint suppression. +- **Why 1 does not fit:** The CI configuration language and almost all values + remain interpretable; the problem is recurring invalid/no-target identifiers, + not the absence of a stable configuration model. +- **Rule discrimination:** Decision rule 6 correctly classifies the no-target + identifiers as domain pressure, but it does not itself distinguish 2 from 3. + The centrality of the API source-of-truth trigger is what selects 2. + +## Control: `fabro-checkpoint` + +### `ownership-boundaries`: 2 (High) + +- **Evidence:** Inside the component, `BranchStore` owns a branch string and + author and delegates Git objects to `Store` + (`src/branch.rs:17-82`), which is a sensible direction. At the production + boundary, however, no production caller constructs `BranchStore`. + `fabro-workflow/src/run_metadata.rs:272-451` instead keeps `Store`, branch, + author, `parent_oid`, and discovery state as separate fields, manually writes + blobs and trees, supplies parents to `Store::write_commit`, resupplies the + branch to `Store::update_ref`, and owns fetch/push discovery. Other checkpoint + commit and trailer lifecycle work also remains in `fabro-workflow`. Thus the + mapped checkpoint/metadata-branch lifecycle crosses the scoped owner on the + normal production path. +- **Strongest counterevidence:** `Store` is itself a mapped public entry point, + the dependency direction remains `fabro-workflow -> fabro-checkpoint`, and + remote authentication/push orchestration reasonably belongs near a workflow + run rather than in a low-level Git object store. +- **Why 3 does not fit:** The caller-held branch and parent identity are used on + every metadata snapshot, not only in an isolated migration or uncommon + fallback. +- **Why 1 does not fit:** Low-level Git ownership and the higher workflow + orchestration are both stable and understandable; they simply split one + routine persistence lifecycle. +- **Rule discrimination:** Decision rule 2 is directly discriminating: + `RunMetadataWriter` retains and repeatedly resupplies the identity needed to + complete operations on `Store`. The mapped breadth of “metadata branches” + makes this more than ordinary parameter passing. + +### `simplicity`: 3 (Medium) + +- **Evidence:** The production low-level path is traceable: + `Store::write_blob -> TreeEntries::set -> Store::write_tree -> + Store::write_commit -> Store::update_ref` + (`src/git.rs:123-188`). `BranchStore::write_with` also gives branch-oriented + writes one linear read/modify/write implementation + (`src/branch.rs:56-117`). The recursive flat-tree conversion is justified by + Git's nested tree representation. The friction is isolated: `BranchStore` is + a sizeable second entry layer with tests but no production caller at this + revision, and `Cargo.toml:18` declares `fabro-store` although scoped + production code does not reference it. +- **Strongest counterevidence:** The two entry points represent legitimate + abstraction levels, and the mapped cartography names both. None of the normal + `Store` operations requires navigating configuration machinery or dynamic + dispatch. +- **Why 4 does not fit:** The unused higher layer/dependency is concrete, + avoidable surface and configuration burden, even though it is off the current + production common path. +- **Why 2 does not fit:** Routine production writes do not repeatedly choose + between `Store` and `BranchStore`; the observed caller consistently uses + `Store`, and that path is direct. +- **Rule discrimination:** The “public method alone does not establish + frequency” rule prevents treating `BranchStore` as a competing common path. + It is less discriminating between 3 and 4; the concrete unused dependency and + unused entry layer are why I select 3. + +### `domain-model`: 2 (High) + +- **Evidence:** `GitAuthor::from_options` accepts arbitrary name/email strings + (`src/author.rs:22-30`), while `BranchStore::new` only interprets them by + calling `Signature::now(...).expect(...)` + (`src/branch.rs:26-39`). `TreeEntries` stores paths as unrestricted `String` + and `BranchStore::write_entry/write_entries` put caller strings into it + without validation (`src/git.rs:46-90`, + `src/branch.rs:84-109`); interpretation and possible rejection occur later + while rebuilding Git trees. `FileMode::from_i32` also maps every unknown Git + mode to `Blob` (`src/git.rs:21-36`) rather than preserving or rejecting an + unknown shape. These invalid-capable intermediates sit on the mapped storage + entry paths. +- **Strongest counterevidence:** `FileMode` is closed for values the component + writes, normal metadata callers validate paths before constructing + `TreeEntries`, Git itself rejects malformed signatures/trees, and object IDs + use git2's typed `Oid`. +- **Why 3 does not fit:** Raw author and path values are carried by the ordinary + entry-point types and interpreted later; they are not confined to a separate + compatibility importer. +- **Why 1 does not fit:** Authors, tree entries, modes, branches, and commits all + have stable intended meanings. The issue is delayed validation and lossy + fallback, not an unidentifiable core concept. +- **Rule discrimination:** Decision rule 4 is discriminating here: these are + common-path invalid-capable intermediate shapes rather than a low-level + escape hatch unused by the entry path. + +### `duplication-knowledge`: 3 (Medium) + +- **Evidence:** Important transformations are mostly authoritative: + `FileMode::{as_i32,from_i32}` contains the mode mapping, + `BranchStore::write_with` contains branch read/modify/write, and + `GitAuthor::default` contains the default identity. The narrow repeated + knowledge is the bare-branch to full-ref transformation + `format!("refs/heads/{branch}")` in each of + `Store::{update_ref,resolve_ref,delete_ref}` + (`src/git.rs:182-225`), with another full-ref rendering at the direct + workflow metadata boundary. Trailer rendering also spells + `"{}: {}"` in both `append` and `format_message` + (`src/trailer.rs:9-65`). +- **Strongest counterevidence:** The repeated ref syntax is stable low-level Git + syntax, the three ref methods implement different operations, and the + apparent duplication in single-entry/multi-entry or tip/commit reads has + intentionally different result shapes. Unifying those operations would risk + a parameterized mega-helper. +- **Why 4 does not fit:** Full-ref and trailer-line rendering have small but real + second representations rather than one helper/type enforcing each + transformation. +- **Why 2 does not fit:** There is no direct evidence of routine changes + repeatedly synchronizing those stable renderings, and hypothetical future ref + methods do not satisfy decision rule 5. The repeated knowledge is isolated + from ordinary checkpoint-format extension. +- **Rule discrimination:** Rule 5 usefully rules out 2 but is + non-discriminating between 3 and 4 for repeated, stable protocol syntax. I + score 3 because the repetitions are concrete, while keeping confidence + Medium because their maintenance materiality is limited. + +## Round 2 revalidation + +I independently reapplied the simplified decision rules to only the requested +assignments. Scores below supersede the corresponding Round 1 judgments for +this revalidation. + +| Component | Lens | Round 2 score | Confidence | +| --- | --- | ---: | --- | +| `fabro-http` | duplication-knowledge | 3 | High | +| `repository-ci` | ownership-boundaries | 2 | High | +| `fabro-checkpoint` | ownership-boundaries | 2 | High | +| `fabro-checkpoint` | simplicity | 3 | High | +| `fabro-checkpoint` | domain-model | 2 | High | +| `fabro-checkpoint` | duplication-knowledge | 3 | Medium | + +### `fabro-http` × `duplication-knowledge`: 3 (High) + +- **Decisive evidence:** `define_builder!` remains the one mechanism for the + materially recurring async/blocking builder policy + (`src/lib.rs:72-193`). The parser and `InvalidProxyPolicy` message still hold + a concrete second representation of the allowed `"system"`/`"disabled"` + vocabulary (`src/lib.rs:29-35,63-69`). +- **Adjacent scores:** 4 does not fit because revised rule 6 explicitly caps a + concrete second semantic representation at 3. Score 2 does not fit because an + ordinary mapped change does not currently synchronize separate async and + blocking implementations; adding a future policy variant is not direct + recurrence evidence. +- **Remaining ambiguity:** None material. Revised rule 6 now resolves the prior + 3-versus-4 uncertainty. + +### `repository-ci` × `ownership-boundaries`: 2 (High) + +- **Decisive evidence:** The Rust workflow's architecture check scans + `apps`, `lib/packages`, and + `docs/public/api-reference/fabro-api.yaml` + (`.github/workflows/rust.yml:80-91`), but its push and pull-request triggers + omit all three routine target paths (`rust.yml:6-19,22-35`). Its Cargo jobs + also consume the real API specification through + `lib/foundation/fabro-api/build.rs`, yet that specification does not trigger + the workflow. The TypeScript workflow likewise consumes the generated API + client and performs the embedded integration build without making the source + specification a trigger. Under revised rule 3, each check owns this coverage; + the omitted routine targets are therefore central ownership pressure. +- **Adjacent scores:** 3 does not fit because API, app, and package changes are + routine targets of checks the workflow actually runs, not isolated edge + inputs. Score 1 does not fit because Rust and TypeScript job ownership and + dependency direction otherwise remain stable. +- **Remaining ambiguity:** None material. The nonexistent `openapi/**` value is + still a separate domain-model finding; the ownership finding rests on the + real scanned/consumed paths that fail to trigger. + +### `fabro-checkpoint` × `ownership-boundaries`: 2 (High) + +- **Decisive evidence:** The mapped higher owner is `BranchStore`, but the + routine production metadata caller instead retains `Store`, branch, author, + parent, and discovery state and reconstructs blob/tree/commit/ref lifecycle + from `Store` primitives in + `fabro-workflow/src/run_metadata.rs:272-451`. Revised rule 2 names this shape + directly. +- **Adjacent scores:** 3 does not fit because reconstruction occurs on every + metadata snapshot, not at an isolated edge. Score 1 does not fit because the + low-level `Store` and workflow-level caller are stable, identifiable owners; + the concern is the lifecycle split between them. +- **Remaining ambiguity:** The workflow reasonably owns remote authentication, + but that does not remove its reconstruction of the mapped checkpoint and + metadata-branch persistence lifecycle. + +### `fabro-checkpoint` × `simplicity`: 3 (High) + +- **Decisive evidence:** The current production `Store` write sequence is + linear and direct (`src/git.rs:123-188`). `BranchStore` is a parallel mapped + entry layer with no production caller at this revision, and `Cargo.toml:18` + declares the unused production dependency `fabro-store`. Revised rule 4 + classifies exactly this as isolated simplicity friction that caps 4 at 3. +- **Adjacent scores:** 4 does not fit because the parallel unused layer and + dependency are concrete. Score 2 does not fit because routine callers do not + navigate competing paths or machinery; they consistently follow the direct + `Store` path. +- **Remaining ambiguity:** None material after rule 4. `BranchStore` being a + mapped entry does not make it frequent when the boundary search finds no + production caller. + +### `fabro-checkpoint` × `domain-model`: 2 (High) + +- **Decisive evidence:** Mapped entry shapes accept unrestricted author and path + strings: `GitAuthor::from_options` stores raw values before + `BranchStore::new` interprets them with `Signature::now(...).expect(...)` + (`src/author.rs:22-30`, `src/branch.rs:26-39`), and + `TreeEntries`/`write_entry` carry unchecked string paths until Git-tree + construction (`src/git.rs:46-90`, `src/branch.rs:84-109`). Revised rule 5 + says caller validation and a typed destination do not isolate this + invalid-capable mapped entry. +- **Adjacent scores:** 3 does not fit because the invalid-capable shapes are on + mapped entry paths, not a compatibility-only edge. Score 1 does not fit + because the intended meanings of authors, paths, modes, and commits remain + stable. +- **Remaining ambiguity:** None material. Normal callers supplying valid values + does not make the entry type canonical by construction. + +### `fabro-checkpoint` × `duplication-knowledge`: 3 (Medium) + +- **Decisive evidence:** Bare branch names are independently rendered as + `refs/heads/{branch}` in `Store::update_ref`, `resolve_ref`, and `delete_ref` + (`src/git.rs:182-225`), and trailer lines are independently rendered in + `trailer::append` and `format_message` (`src/trailer.rs:9-65`). These are + concrete second semantic representations, so revised rule 6 excludes 4. +- **Adjacent scores:** 4 does not fit because the second renderings are real. + Score 2 does not fit because no evidenced ordinary mapped change must + synchronize the stable Git ref or trailer syntax across those locations; + future ref operations are hypothetical, while the existing operations have + distinct behavior. +- **Remaining ambiguity:** Limited ambiguity remains over whether stable + protocol syntax is material enough to count as semantic repetition at all. + Rule 6 does not define that threshold, so confidence remains Medium; if it + counts, 3 is the rule-directed score. From 59b1c2e59ffca7cf5f491c4fe65e7650afb48a28 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 13:53:54 -0400 Subject: [PATCH 13/76] Reject nodes referenced by an edge but never declared MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The DOT parser created a node for every edge endpoint, and nothing recorded whether a node came from a declaration or was synthesized from an edge. The edge_target_exists rule only checked whether the node id was present in the graph, which was always true by then, so a misspelled endpoint became an attribute-free node that defaulted to shape=box — an LLM stage. Validation emitted a prompt_on_llm_nodes warning and exited 0. Node now carries `implicit`, set only when the parser synthesizes the node from an edge endpoint. A declaration anywhere in the workflow clears it, so order does not matter and subgraph declarations count. Node::new leaves it false, so programmatic construction and graphs deserialized from older checkpoints read as declared. edge_target_exists treats an endpoint as valid only when it exists and is declared, reporting each undeclared node once. The near-identical missing-source and missing-target branches collapse into one path. The import transform copies the flag onto spliced nodes so an edge-only node inside an imported fragment is caught too. parse_and_validate_human_gate had two edge-only nodes and now declares them; it was an instance of the bug rather than a casualty of the fix. No shipped workflow, docs example, or CLI fixture relied on the old behavior. Co-Authored-By: Claude Opus 5 (1M context) --- docs/public/reference/dot-language.mdx | 2 +- lib/apps/fabro-cli/tests/it/cmd/validate.rs | 20 +++ .../fabro-graphviz/src/parser/semantic.rs | 93 +++++++++- .../src/rules/edge_target_exists.rs | 168 ++++++++++++++---- .../fabro-workflow/src/transforms/import.rs | 49 +++++ .../fabro-workflow/tests/it/integration.rs | 3 + lib/foundation/fabro-types/src/graph.rs | 18 +- .../fabro-types/src/run_event/mod.rs | 7 +- test/edge_only_node.fabro | 10 ++ 9 files changed, 326 insertions(+), 44 deletions(-) create mode 100644 test/edge_only_node.fabro diff --git a/docs/public/reference/dot-language.mdx b/docs/public/reference/dot-language.mdx index 8c350d1ba..ee44c3668 100644 --- a/docs/public/reference/dot-language.mdx +++ b/docs/public/reference/dot-language.mdx @@ -113,7 +113,7 @@ plan [label="Plan", prompt="Create an implementation plan."] **Node identifiers** must start with a letter or underscore, followed by letters, digits, or underscores (e.g. `run_tests`, `gate_1`, `_private`). -Nodes referenced in edges are auto-created if not explicitly declared. +Every node used by an edge needs its own declaration. Validation fails when an edge names a node the workflow never declares, because that is nearly always a typo or a rename that missed an edge. The declaration can come before or after the edges that use it, and it can live in a subgraph. ### Edge declarations diff --git a/lib/apps/fabro-cli/tests/it/cmd/validate.rs b/lib/apps/fabro-cli/tests/it/cmd/validate.rs index 190c2817b..98b0f9cfe 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/validate.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/validate.rs @@ -278,6 +278,26 @@ fn validate_reports_missing_template_dependency() { "); } +/// A node named only by an edge is almost always a typo, so validation must +/// fail instead of quietly running it as a default agent stage. +#[test] +fn edge_only_node() { + let context = test_context!(); + let mut cmd = context.validate(); + cmd.arg(fixture("edge_only_node.fabro")); + fabro_snapshot!(context.filters(), cmd, @" + success: false + exit_code: 1 + ----- stdout ----- + ----- stderr ----- + Workflow: EdgeOnlyNode (3 nodes, 2 edges) + Graph: [FIXTURES]/edge_only_node.fabro + error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists) + warning [node: misspelled_node]: LLM node 'misspelled_node' has no prompt or label attribute (prompt_on_llm_nodes) + × Validation failed + "); +} + #[test] fn invalid() { let context = test_context!(); diff --git a/lib/components/fabro-graphviz/src/parser/semantic.rs b/lib/components/fabro-graphviz/src/parser/semantic.rs index 8ac364cab..4e9433e6b 100644 --- a/lib/components/fabro-graphviz/src/parser/semantic.rs +++ b/lib/components/fabro-graphviz/src/parser/semantic.rs @@ -57,6 +57,14 @@ fn derive_class_from_label(label: &str) -> String { .collect() } +/// How a statement named a node. A node stays implicit only while every +/// mention of it is an edge endpoint, so declaration order does not matter. +#[derive(Clone, Copy, PartialEq, Eq)] +enum Mention { + Declaration, + EdgeEndpoint, +} + struct SemanticState { graph: Graph, node_defaults: HashMap, @@ -72,13 +80,25 @@ impl SemanticState { } } - fn ensure_node(&mut self, id: &str) { + /// Insert the node if this is the first statement to mention it, and record + /// whether the workflow ever declares it. + fn ensure_node(&mut self, id: &str, mention: Mention) { if !self.graph.nodes.contains_key(id) { let mut node = Node::new(id); for (k, v) in &self.node_defaults { node.attrs.insert(k.clone(), v.clone()); } + node.implicit = mention == Mention::EdgeEndpoint; self.graph.nodes.insert(id.to_string(), node); + return; + } + if mention == Mention::Declaration { + let node = self + .graph + .nodes + .get_mut(id) + .expect("contains_key returned true, so get_mut cannot return None"); + node.implicit = false; } } @@ -90,7 +110,7 @@ impl SemanticState { } fn process_node(&mut self, node_stmt: &NodeStmt, subgraph_class: Option<&str>) { - self.ensure_node(&node_stmt.id); + self.ensure_node(&node_stmt.id, Mention::Declaration); let node = self .graph .nodes @@ -127,7 +147,7 @@ impl SemanticState { fn process_edge(&mut self, edge_stmt: &EdgeStmt, subgraph_class: Option<&str>) { for id in &edge_stmt.nodes { - self.ensure_node(id); + self.ensure_node(id, Mention::EdgeEndpoint); if let Some(cls) = subgraph_class { let node = self.graph.nodes.get_mut(id).expect( @@ -527,5 +547,72 @@ mod tests { let graph = ast_to_graph(&dot).unwrap(); assert!(graph.nodes.contains_key("a")); assert!(graph.nodes.contains_key("b")); + assert!(graph.nodes["a"].implicit); + assert!(graph.nodes["b"].implicit); + } + + #[test] + fn ast_to_graph_marks_declared_nodes_explicit() { + let dot = DotGraph { + name: "Declared".into(), + statements: vec![ + Statement::Node(NodeStmt { + id: "a".into(), + attrs: None, + }), + Statement::Edge(EdgeStmt { + nodes: vec!["a".into(), "b".into()], + attrs: None, + }), + ], + }; + + let graph = ast_to_graph(&dot).unwrap(); + assert!(!graph.nodes["a"].implicit); + assert!(graph.nodes["b"].implicit); + } + + #[test] + fn ast_to_graph_declaration_after_edge_still_counts() { + let dot = DotGraph { + name: "DeclaredLater".into(), + statements: vec![ + Statement::Edge(EdgeStmt { + nodes: vec!["a".into(), "b".into()], + attrs: None, + }), + Statement::Node(NodeStmt { + id: "b".into(), + attrs: Some(vec![("prompt".into(), AstValue::Str("Do it".into()))]), + }), + ], + }; + + let graph = ast_to_graph(&dot).unwrap(); + assert!(!graph.nodes["b"].implicit); + } + + #[test] + fn ast_to_graph_subgraph_declaration_counts() { + let dot = DotGraph { + name: "SubgraphDeclared".into(), + statements: vec![ + Statement::Edge(EdgeStmt { + nodes: vec!["start".into(), "plan".into()], + attrs: None, + }), + Statement::Subgraph(SubgraphStmt { + name: Some("cluster_loop".into()), + statements: vec![Statement::Node(NodeStmt { + id: "plan".into(), + attrs: None, + })], + }), + ], + }; + + let graph = ast_to_graph(&dot).unwrap(); + assert!(!graph.nodes["plan"].implicit); + assert!(graph.nodes["start"].implicit); } } diff --git a/lib/components/fabro-validate/src/rules/edge_target_exists.rs b/lib/components/fabro-validate/src/rules/edge_target_exists.rs index 8cfe67282..adc70283b 100644 --- a/lib/components/fabro-validate/src/rules/edge_target_exists.rs +++ b/lib/components/fabro-validate/src/rules/edge_target_exists.rs @@ -1,3 +1,5 @@ +use std::collections::HashSet; + use fabro_graphviz::graph::Graph; use crate::{Diagnostic, LintRule, Severity}; @@ -8,6 +10,33 @@ pub(super) fn rule() -> Box { struct Rule; +impl Rule { + /// An edge endpoint is only usable when the workflow declares it. A node + /// the parser synthesized from the edge itself carries no attributes, so it + /// would silently run as a default agent stage. + fn is_declared(graph: &Graph, node_id: &str) -> bool { + graph.nodes.get(node_id).is_some_and(|node| !node.implicit) + } + + fn diagnostic(&self, node_id: &str, from: &str, to: &str) -> Diagnostic { + Diagnostic { + rule: self.name().to_string(), + severity: Severity::Error, + message: format!( + "Node '{node_id}' is referenced by edge '{from} -> {to}' but has no node \ + declaration" + ), + node_id: Some(node_id.to_string()), + edge: Some((from.to_string(), to.to_string())), + fix: Some(format!( + "Declare node '{node_id}' or correct the edge endpoint" + )), + + ..Diagnostic::default() + } + } +} + impl LintRule for Rule { fn name(&self) -> &'static str { "edge_target_exists" @@ -15,36 +44,12 @@ impl LintRule for Rule { fn apply(&self, graph: &Graph) -> Vec { let mut diagnostics = Vec::new(); + let mut reported = HashSet::new(); for edge in &graph.edges { - if !graph.nodes.contains_key(&edge.to) { - diagnostics.push(Diagnostic { - rule: self.name().to_string(), - severity: Severity::Error, - message: format!( - "Edge from '{}' targets non-existent node '{}'", - edge.from, edge.to - ), - node_id: None, - edge: Some((edge.from.clone(), edge.to.clone())), - fix: Some(format!("Define node '{}' or fix the edge target", edge.to)), - - ..Diagnostic::default() - }); - } - if !graph.nodes.contains_key(&edge.from) { - diagnostics.push(Diagnostic { - rule: self.name().to_string(), - severity: Severity::Error, - message: format!("Edge source '{}' references non-existent node", edge.from), - node_id: None, - edge: Some((edge.from.clone(), edge.to.clone())), - fix: Some(format!( - "Define node '{}' or fix the edge source", - edge.from - )), - - ..Diagnostic::default() - }); + for endpoint in [&edge.to, &edge.from] { + if !Self::is_declared(graph, endpoint) && reported.insert(endpoint) { + diagnostics.push(self.diagnostic(endpoint, &edge.from, &edge.to)); + } } } diagnostics @@ -53,11 +58,112 @@ impl LintRule for Rule { #[cfg(test)] mod tests { - use fabro_graphviz::graph::Edge; + use fabro_graphviz::graph::{Edge, Graph}; + use fabro_graphviz::parser; use super::Rule; use crate::rules::test_support::minimal_graph; - use crate::{LintRule, Severity}; + use crate::{Diagnostic, LintRule, Severity}; + + fn parse(dot: &str) -> Graph { + parser::parse(dot).expect("fixture should parse") + } + + fn undeclared_nodes(graph: &Graph) -> Vec { + Rule.apply(graph) + .iter() + .map(|d| d.node_id.clone().expect("diagnostic should name a node")) + .collect() + } + + #[test] + fn edge_only_node_is_rejected() { + let graph = parse( + r"digraph EdgeOnly { + start [shape=Mdiamond] + exit [shape=Msquare] + start -> misspelled_node + misspelled_node -> exit + }", + ); + + let diagnostics = Rule.apply(&graph); + assert_eq!(diagnostics.len(), 1, "diagnostics: {diagnostics:?}"); + let Diagnostic { + severity, + node_id, + edge, + .. + } = &diagnostics[0]; + assert_eq!(*severity, Severity::Error); + assert_eq!(node_id.as_deref(), Some("misspelled_node")); + assert_eq!( + edge.clone(), + Some(("start".to_string(), "misspelled_node".to_string())) + ); + } + + #[test] + fn declaration_after_the_edge_is_accepted() { + let graph = parse( + r#"digraph DeclaredLater { + start -> work + work [prompt="Do the work"] + work -> exit + start [shape=Mdiamond] + exit [shape=Msquare] + }"#, + ); + + assert!(Rule.apply(&graph).is_empty()); + } + + #[test] + fn chained_edges_report_every_undeclared_endpoint() { + let graph = parse( + r"digraph Chained { + start [shape=Mdiamond] + exit [shape=Msquare] + start -> first -> second -> exit + }", + ); + + assert_eq!(undeclared_nodes(&graph), vec!["first", "second"]); + } + + #[test] + fn a_node_is_reported_once_no_matter_how_many_edges_use_it() { + let graph = parse( + r"digraph Repeated { + start [shape=Mdiamond] + exit [shape=Msquare] + start -> typo + typo -> exit + typo -> start + }", + ); + + assert_eq!(undeclared_nodes(&graph), vec!["typo"]); + } + + #[test] + fn subgraph_declaration_is_accepted() { + let graph = parse( + r#"digraph Subgraphed { + start [shape=Mdiamond] + exit [shape=Msquare] + + subgraph cluster_loop { + label = "Loop A" + plan [prompt="Plan the work"] + } + + start -> plan -> exit + }"#, + ); + + assert!(Rule.apply(&graph).is_empty()); + } #[test] fn edge_target_exists_rule_missing_target() { diff --git a/lib/components/fabro-workflow/src/transforms/import.rs b/lib/components/fabro-workflow/src/transforms/import.rs index b5406ddc6..cfda146c3 100644 --- a/lib/components/fabro-workflow/src/transforms/import.rs +++ b/lib/components/fabro-workflow/src/transforms/import.rs @@ -346,6 +346,7 @@ impl ImportTransform { let prefixed_id = format!("{placeholder_id}.{node_id}"); let mut merged_node = Node::new(&prefixed_id); + merged_node.implicit = node.implicit; merged_node.attrs.clone_from(&placeholder.default_attrs); merged_node.attrs.extend(node.attrs); Self::remap_retry_target(&mut merged_node.attrs, placeholder_id); @@ -951,6 +952,54 @@ mod tests { ); } + #[test] + fn imported_node_declarations_survive_splicing() { + let dir = tempfile::tempdir().unwrap(); + write_file(&dir.path().join("validate.fabro"), basic_import_source()); + + let graph = apply_import( + r#"digraph Deploy { + start [shape=Mdiamond] + validate [import="./validate.fabro"] + exit [shape=Msquare] + start -> validate -> exit + }"#, + dir.path(), + None, + ); + + assert!(!graph.nodes["validate.lint"].implicit); + assert!(!graph.nodes["validate.test"].implicit); + } + + #[test] + fn edge_only_node_in_imported_fragment_stays_undeclared() { + let dir = tempfile::tempdir().unwrap(); + write_file( + &dir.path().join("validate.fabro"), + r#"digraph validate { + start [shape=Mdiamond] + lint [prompt="Run clippy"] + exit [shape=Msquare] + start -> lint -> typo -> exit + }"#, + ); + + let graph = apply_import( + r#"digraph Deploy { + start [shape=Mdiamond] + validate [import="./validate.fabro"] + exit [shape=Msquare] + start -> validate -> exit + }"#, + dir.path(), + None, + ); + + assert!(graph.nodes["validate.typo"].implicit); + assert!(!graph.nodes["validate.lint"].implicit); + } + #[test] fn import_reports_structural_diagnostic_for_imported_prompt_templates() { let dir = tempfile::tempdir().unwrap(); diff --git a/lib/components/fabro-workflow/tests/it/integration.rs b/lib/components/fabro-workflow/tests/it/integration.rs index dc13bcbf8..ec716adbc 100644 --- a/lib/components/fabro-workflow/tests/it/integration.rs +++ b/lib/components/fabro-workflow/tests/it/integration.rs @@ -388,6 +388,9 @@ fn parse_and_validate_human_gate() { type="human" ] + ship_it [prompt="Ship the change"] + fixes [prompt="Apply the requested fixes"] + start -> review_gate review_gate -> ship_it [label="[A] Approve"] review_gate -> fixes [label="[F] Fix"] diff --git a/lib/foundation/fabro-types/src/graph.rs b/lib/foundation/fabro-types/src/graph.rs index 7ef9bad99..3ed6851a6 100644 --- a/lib/foundation/fabro-types/src/graph.rs +++ b/lib/foundation/fabro-types/src/graph.rs @@ -119,20 +119,26 @@ pub fn shape_to_handler_type(shape: &str) -> Option<&'static str> { /// A node in the workflow graph. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct Node { - pub id: String, - pub attrs: HashMap, + pub id: String, + pub attrs: HashMap, /// CSS-like classes for model stylesheet targeting (from `class` attr and /// subgraph derivation). #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub classes: Vec, + pub classes: Vec, + /// True when the node was synthesized from an edge endpoint instead of a + /// node declaration. Validation rejects these because an edge-only node in + /// an executable workflow is almost always a typo. + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + pub implicit: bool, } impl Node { pub fn new(id: impl Into) -> Self { Self { - id: id.into(), - attrs: HashMap::new(), - classes: Vec::new(), + id: id.into(), + attrs: HashMap::new(), + classes: Vec::new(), + implicit: false, } } diff --git a/lib/foundation/fabro-types/src/run_event/mod.rs b/lib/foundation/fabro-types/src/run_event/mod.rs index d4c8e55c1..e4e99e59f 100644 --- a/lib/foundation/fabro-types/src/run_event/mod.rs +++ b/lib/foundation/fabro-types/src/run_event/mod.rs @@ -995,9 +995,10 @@ mod tests { let graph = Graph { name: "test".to_string(), nodes: HashMap::from([("start".to_string(), Node { - id: "start".to_string(), - attrs: HashMap::new(), - classes: Vec::new(), + id: "start".to_string(), + attrs: HashMap::new(), + classes: Vec::new(), + implicit: false, })]), edges: vec![Edge { from: "start".to_string(), diff --git a/test/edge_only_node.fabro b/test/edge_only_node.fabro new file mode 100644 index 000000000..3d45b3346 --- /dev/null +++ b/test/edge_only_node.fabro @@ -0,0 +1,10 @@ +digraph EdgeOnlyNode { + graph [goal="Reference a node that was never declared"] + + /* `misspelled_node` is only ever named by an edge, never declared. */ + start [shape=Mdiamond, label="Start"] + exit [shape=Msquare, label="Exit"] + + start -> misspelled_node + misspelled_node -> exit +} From c501c67185892cad0714859eedb4676e1552d7a6 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 14:06:31 -0400 Subject: [PATCH 14/76] Show each diagnostic's suggested fix in CLI output MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Diagnostics have carried a `fix` field all along, but the CLI renderer never printed it — the suggestion was only reachable through --json. The actionable half of every validation failure was invisible to the person running the command. print_diagnostics now emits the fix as a dim-labelled continuation line under any diagnostic that has one, at both error and warning severity. Gating it behind --verbose would defeat the point, and printing it only for errors would read as "this warning has no fix" — the warning suggestions are useful on their own. Diagnostics that set no fix simply omit the line. The severity match moved into print_diagnostic so the fix line is appended once in the loop rather than copied into all five arms; the rest of the diff is reindentation. print_diagnostics is shared by validate, preflight, graph, exec, and dry-run, so this covers all five. Eleven inline snapshots across four files gain a fix line; every change is additive. Co-Authored-By: Claude Opus 5 (1M context) --- lib/apps/fabro-cli/src/shared/utilities.rs | 98 ++++++++++--------- lib/apps/fabro-cli/tests/it/cmd/graph.rs | 4 + lib/apps/fabro-cli/tests/it/cmd/preflight.rs | 4 + lib/apps/fabro-cli/tests/it/cmd/validate.rs | 9 ++ .../tests/it/workflow/dry_run_examples.rs | 1 + 5 files changed, 72 insertions(+), 44 deletions(-) diff --git a/lib/apps/fabro-cli/src/shared/utilities.rs b/lib/apps/fabro-cli/src/shared/utilities.rs index e9cd32bf6..8cfb05af7 100644 --- a/lib/apps/fabro-cli/src/shared/utilities.rs +++ b/lib/apps/fabro-cli/src/shared/utilities.rs @@ -49,54 +49,64 @@ where pub(crate) fn print_diagnostics(diagnostics: &[Diagnostic], styles: &Styles, printer: Printer) { for d in diagnostics { - let location = match (&d.node_id, &d.edge) { - (Some(node), _) => format!(" [node: {node}]"), - (_, Some((from, to))) => format!(" [edge: {from} -> {to}]"), - _ => String::new(), - }; - let source_prefix = source_prefix(d); - match d.severity { - Severity::Error if source_prefix.is_empty() => fabro_util::printerr!( - printer, - "{}{location}: {} ({})", - styles.red.apply_to("error"), - d.message, - styles.dim.apply_to(&d.rule), - ), - Severity::Error => fabro_util::printerr!( - printer, - "{}: {source_prefix}{}{location} ({})", - styles.red.apply_to("error"), - d.message, - styles.dim.apply_to(&d.rule), - ), - Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!( - printer, - "{}{location}: {} ({})", - styles.yellow.apply_to("warning"), - d.message, - styles.dim.apply_to(&d.rule), - ), - Severity::Warning => fabro_util::printerr!( - printer, - "{}: {source_prefix}{}{location} ({})", - styles.yellow.apply_to("warning"), - d.message, - styles.dim.apply_to(&d.rule), - ), - Severity::Info => fabro_util::printerr!( - printer, - "{}", - styles.dim.apply_to(if source_prefix.is_empty() { - format!("info{location}: {} ({})", d.message, d.rule) - } else { - format!("info: {source_prefix}{}{location} ({})", d.message, d.rule) - }), - ), + print_diagnostic(d, styles, printer); + // The fix is the actionable half of a diagnostic, so it follows every + // severity rather than hiding behind --verbose. Rules that have nothing + // useful to suggest leave it unset. + if let Some(fix) = &d.fix { + fabro_util::printerr!(printer, " {} {fix}", styles.dim.apply_to("fix:")); } } } +fn print_diagnostic(d: &Diagnostic, styles: &Styles, printer: Printer) { + let location = match (&d.node_id, &d.edge) { + (Some(node), _) => format!(" [node: {node}]"), + (_, Some((from, to))) => format!(" [edge: {from} -> {to}]"), + _ => String::new(), + }; + let source_prefix = source_prefix(d); + match d.severity { + Severity::Error if source_prefix.is_empty() => fabro_util::printerr!( + printer, + "{}{location}: {} ({})", + styles.red.apply_to("error"), + d.message, + styles.dim.apply_to(&d.rule), + ), + Severity::Error => fabro_util::printerr!( + printer, + "{}: {source_prefix}{}{location} ({})", + styles.red.apply_to("error"), + d.message, + styles.dim.apply_to(&d.rule), + ), + Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!( + printer, + "{}{location}: {} ({})", + styles.yellow.apply_to("warning"), + d.message, + styles.dim.apply_to(&d.rule), + ), + Severity::Warning => fabro_util::printerr!( + printer, + "{}: {source_prefix}{}{location} ({})", + styles.yellow.apply_to("warning"), + d.message, + styles.dim.apply_to(&d.rule), + ), + Severity::Info => fabro_util::printerr!( + printer, + "{}", + styles.dim.apply_to(if source_prefix.is_empty() { + format!("info{location}: {} ({})", d.message, d.rule) + } else { + format!("info: {source_prefix}{}{location} ({})", d.message, d.rule) + }), + ), + } +} + fn source_prefix(diagnostic: &Diagnostic) -> String { match ( diagnostic.source_path.as_deref(), diff --git a/lib/apps/fabro-cli/tests/it/cmd/graph.rs b/lib/apps/fabro-cli/tests/it/cmd/graph.rs index 6b801928a..fbf48599f 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/graph.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/graph.rs @@ -95,7 +95,9 @@ fn graph_allow_invalid_renders_after_diagnostics() { ----- stdout ----- ----- stderr ----- error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node) + fix: Add a node with shape=Mdiamond or id 'start' error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing) + fix: Remove outgoing edges from the exit node "); let svg = read_text(&output_path); @@ -119,7 +121,9 @@ fn graph_invalid_workflow_fails_after_diagnostics() { ----- stdout ----- ----- stderr ----- error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node) + fix: Add a node with shape=Mdiamond or id 'start' error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing) + fix: Remove outgoing edges from the exit node × Validation failed "); } diff --git a/lib/apps/fabro-cli/tests/it/cmd/preflight.rs b/lib/apps/fabro-cli/tests/it/cmd/preflight.rs index 8bb035f3a..4a2e44c8a 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/preflight.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/preflight.rs @@ -52,7 +52,9 @@ fn preflight_invalid_workflow_fails_with_validation_output() { Workflow: Invalid (2 nodes, 1 edges) Graph: [FIXTURES]/invalid.fabro error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node) + fix: Add a node with shape=Mdiamond or id 'start' error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing) + fix: Remove outgoing edges from the exit node × Validation failed "); } @@ -74,7 +76,9 @@ fn preflight_rejects_unbound_template_inputs() { Goal: Demo error: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable) + fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=` error: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable) + fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=` × Validation failed "); } diff --git a/lib/apps/fabro-cli/tests/it/cmd/validate.rs b/lib/apps/fabro-cli/tests/it/cmd/validate.rs index 98b0f9cfe..caeab11ca 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/validate.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/validate.rs @@ -82,6 +82,7 @@ fn branching() { Workflow: Branch (6 nodes, 6 edges) Graph: [FIXTURES]/branching.fabro warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry) + fix: Add retry_target or fallback_retry_target attribute Validation: OK "); } @@ -163,7 +164,9 @@ fn bare_fabro_with_unbound_inputs_validates_structurally_with_warning() { Workflow: TemplatedUnbound (3 nodes, 2 edges) Graph: [FIXTURES]/templated_unbound.fabro warning: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable) + fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=` warning: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable) + fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=` Validation: OK "); } @@ -186,6 +189,7 @@ fn bare_fabro_with_unbound_inputs_in_imported_prompt_validates_structurally_with Workflow: TemplatedUnboundImported (3 nodes, 2 edges) Graph: [FIXTURES]/templated_unbound_imported/workflow.fabro warning: [FIXTURES]/templated_unbound_imported/work.md:1:12: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable) + fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=` Validation: OK "); } @@ -207,6 +211,7 @@ fn bare_fabro_with_unbound_inputs_in_template_partial_validates_structurally_wit Workflow: TemplatedUnboundPartial (3 nodes, 2 edges) Graph: [FIXTURES]/templated_unbound_partial/workflow.fabro warning: [FIXTURES]/templated_unbound_partial/test-include.partial.md:1:4: undefined template variable `inputs.hello` in node `test_imported_include` attribute `prompt` [node: test_imported_include] (template_undefined_variable) + fix: bind `inputs.hello` via `[run.inputs]` in workflow.toml, or pass `--input inputs.hello=` Validation: OK "); } @@ -293,7 +298,9 @@ fn edge_only_node() { Workflow: EdgeOnlyNode (3 nodes, 2 edges) Graph: [FIXTURES]/edge_only_node.fabro error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists) + fix: Declare node 'misspelled_node' or correct the edge endpoint warning [node: misspelled_node]: LLM node 'misspelled_node' has no prompt or label attribute (prompt_on_llm_nodes) + fix: Add a prompt or label attribute × Validation failed "); } @@ -311,7 +318,9 @@ fn invalid() { Workflow: Invalid (2 nodes, 1 edges) Graph: [FIXTURES]/invalid.fabro error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node) + fix: Add a node with shape=Mdiamond or id 'start' error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing) + fix: Remove outgoing edges from the exit node × Validation failed "); } diff --git a/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs b/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs index 658dd1fea..d6852765e 100644 --- a/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs +++ b/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs @@ -19,6 +19,7 @@ fn dry_run_branching() { Goal: Implement and validate a feature warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry) + fix: Add retry_target or fallback_retry_target attribute Run: [ULID] Web UI: http://localhost:3000/runs/[ULID] Sandbox: local (ready in [TIME]) From 716ba1778069f5f916f34515de79a1f0c417511e Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 15:19:57 -0400 Subject: [PATCH 15/76] Show billed amount in runs list size tooltip MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Size column in the runs list rendered SizeChip without the billed total, so its tooltip read "Size M" while the run detail header showed "Size M · $12.34 billed". The tooltip was also unreachable: the row title link paints a `before:absolute before:inset-0` overlay across the whole row, which sat above the chip and swallowed hover. Wrapping the chip in `relative z-10` lifts it above that overlay, matching how the created-by and pull request cells already handle interactive content. Runs without terminal billing keep the plain "Size M" label, same as the header. Co-Authored-By: Claude Opus 5 (1M context) --- .../app/components/runs-list/run-table-row.tsx | 6 +++++- apps/fabro-web/app/data/runs.test.ts | 11 +++++++++++ apps/fabro-web/app/data/runs.ts | 2 ++ 3 files changed, 18 insertions(+), 1 deletion(-) diff --git a/apps/fabro-web/app/components/runs-list/run-table-row.tsx b/apps/fabro-web/app/components/runs-list/run-table-row.tsx index 850c36467..e7447bce5 100644 --- a/apps/fabro-web/app/components/runs-list/run-table-row.tsx +++ b/apps/fabro-web/app/components/runs-list/run-table-row.tsx @@ -116,7 +116,11 @@ export function RunTableRow({ )} {show("size") && ( - {run.size != null && } + {run.size != null && ( + + + + )} )} {show("changes") && ( diff --git a/apps/fabro-web/app/data/runs.test.ts b/apps/fabro-web/app/data/runs.test.ts index 0ad593526..77fdc91e2 100644 --- a/apps/fabro-web/app/data/runs.test.ts +++ b/apps/fabro-web/app/data/runs.test.ts @@ -103,6 +103,17 @@ describe("mapRunListItem", () => { expect(mapRunListItem(summary).title).toBe("Untitled run"); }); + + test("carries the billed total so the size chip can show it on hover", () => { + expect(mapRunListItem(makeRun()).totalUsdMicros).toBe(500000); + }); + + test("leaves the billed total undefined for runs without terminal billing", () => { + expect(mapRunListItem(makeRun({ billing: null })).totalUsdMicros).toBeUndefined(); + expect( + mapRunListItem(makeRun({ billing: { total_usd_micros: null } })).totalUsdMicros, + ).toBeUndefined(); + }); }); describe("mapRunToRunItem", () => { diff --git a/apps/fabro-web/app/data/runs.ts b/apps/fabro-web/app/data/runs.ts index 4fb704133..c5f29f6ae 100644 --- a/apps/fabro-web/app/data/runs.ts +++ b/apps/fabro-web/app/data/runs.ts @@ -46,6 +46,7 @@ export interface RunItem { createdBy: Principal; lastEventAt?: string; size?: RunSize; + totalUsdMicros?: number; } export const columnStatuses = [ @@ -119,6 +120,7 @@ export function mapRunListItem(item: Run): RunItem { additions: item.diff?.additions, deletions: item.diff?.deletions, size: item.size, + totalUsdMicros: item.billing?.total_usd_micros ?? undefined, }; } From 991f160a0b09f54f931fe27813ccc011779073ec Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 15:23:01 -0400 Subject: [PATCH 16/76] Drop "billed" from the size chip tooltip MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The tooltip now reads "Size M · $12.34" instead of "Size M · $12.34 billed". Co-Authored-By: Claude Opus 5 (1M context) --- apps/fabro-web/app/components/size-chip.tsx | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/apps/fabro-web/app/components/size-chip.tsx b/apps/fabro-web/app/components/size-chip.tsx index e3d07ed40..d23210955 100644 --- a/apps/fabro-web/app/components/size-chip.tsx +++ b/apps/fabro-web/app/components/size-chip.tsx @@ -19,10 +19,10 @@ export function SizeChip({ totalUsdMicros?: number | null; }) { const tone = SIZE_TONE[size]; - const billed = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)} billed` : ""; + const amount = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)}` : ""; const tooltip = tone.note != null - ? `Size ${size} (${tone.note})${billed}` - : `Size ${size}${billed}`; + ? `Size ${size} (${tone.note})${amount}` + : `Size ${size}${amount}`; return ( From 53c580ce5fde89d4ef348bf091da8074c5a66aee Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 15:25:44 -0400 Subject: [PATCH 17/76] Swap the board card's elapsed time for a size chip The board cards showed wall-clock duration in the footer's bottom-right corner. Replace it with the same SizeChip the list view and run detail header use, so the cost signal is consistent across all three views. The chip inherits the tooltip, which names the tier and adds the cost once a run has terminal billing. Add SizeChip tests pinning the tooltip label for each tier. Co-Authored-By: Claude Opus 5 (1M context) --- .../app/components/size-chip.test.tsx | 40 +++++++++++++++++++ apps/fabro-web/app/routes/runs.tsx | 11 ++--- 2 files changed, 46 insertions(+), 5 deletions(-) create mode 100644 apps/fabro-web/app/components/size-chip.test.tsx diff --git a/apps/fabro-web/app/components/size-chip.test.tsx b/apps/fabro-web/app/components/size-chip.test.tsx new file mode 100644 index 000000000..b355f5db1 --- /dev/null +++ b/apps/fabro-web/app/components/size-chip.test.tsx @@ -0,0 +1,40 @@ +import { describe, expect, test } from "bun:test"; +import TestRenderer, { act } from "react-test-renderer"; + +import { SizeChip } from "./size-chip"; +import { Tooltip } from "./ui"; + +function tooltipLabel(element: React.ReactElement): string { + let renderer: TestRenderer.ReactTestRenderer | undefined; + act(() => { + renderer = TestRenderer.create(element); + }); + return renderer!.root.findByType(Tooltip).props.label as string; +} + +describe("SizeChip", () => { + test("renders the size letter", () => { + let renderer: TestRenderer.ReactTestRenderer | undefined; + act(() => { + renderer = TestRenderer.create(); + }); + + expect(JSON.stringify(renderer!.toJSON())).toContain("M"); + }); + + test("appends the cost to the tooltip", () => { + expect(tooltipLabel()) + .toBe("Size M · $12.34"); + }); + + test("omits the cost when the run has no billing yet", () => { + expect(tooltipLabel()).toBe("Size M"); + expect(tooltipLabel()).toBe("Size M"); + }); + + test("calls out the tiers that warrant attention", () => { + expect(tooltipLabel()) + .toBe("Size L (risky) · $150.00"); + expect(tooltipLabel()).toBe("Size XL (unhealthy)"); + }); +}); diff --git a/apps/fabro-web/app/routes/runs.tsx b/apps/fabro-web/app/routes/runs.tsx index 7da0719ea..fe21e36c8 100644 --- a/apps/fabro-web/app/routes/runs.tsx +++ b/apps/fabro-web/app/routes/runs.tsx @@ -26,6 +26,7 @@ import { ciConfig, columnForRun, columnStatusDisplay, columnStatuses, deriveCiSt import type { CiStatus, CheckRun, CheckStatus, RunItem } from "../data/runs"; import { EmptyState } from "../components/state"; import { PullRequestChip } from "../components/pull-request-chip"; +import { SizeChip } from "../components/size-chip"; import { summarizeBatchLifecycleAction, } from "../components/runs-list/batch-lifecycle"; @@ -345,7 +346,7 @@ function PrCard({ // All inline footer metadata on PrCard belongs in this one row. Adding a new // piece as a sibling `
` below the card body recreates a recurring bug -// where stats stack onto separate lines instead of sitting next to elapsed/actions. +// where stats stack onto separate lines instead of sitting next to size/actions. function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) { const hasActions = actions != null && actions.length > 0; const hasStats = @@ -354,7 +355,7 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) { (pr.additions != null && pr.additions !== 0) || (pr.deletions != null && pr.deletions !== 0); - if (!hasStats && !hasActions && pr.elapsed == null) return null; + if (!hasStats && !hasActions && pr.size == null) return null; return (
@@ -416,9 +417,9 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) { ))}
)} - {pr.elapsed != null && ( - - {pr.elapsed} + {pr.size != null && ( + + )}
From 7841a77f2c0962439864ebdf8f04ba9441fba7fd Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Mon, 27 Jul 2026 16:06:19 -0400 Subject: [PATCH 18/76] feat(web): show stage tokens and cost in the model popover The model indicator on a stage page hovered to provider, model, and reasoning effort only. Seeing what a stage actually spent meant leaving for the Billing tab, which reports per node rather than per visit. The stage list had no token data to show, so add a per-visit `billing` block to `GET /runs/{id}/stages`. The Billing tab's pricing rule (a provider-reported cost wins, otherwise the server catalog prices the tokens) was private to `billing_rollup`; move it to `StageProjection::billed_usage` and drive both call sites from it so the two views cannot drift. The popover's buckets use the Billing tab's labels verbatim. It stays scoped to one visit, so a looped node's row on the Billing tab is the sum of what each of its visits shows here. Co-Authored-By: Claude Opus 5 (1M context) --- apps/fabro-web/app/lib/stage-sidebar.test.ts | 22 +++ apps/fabro-web/app/lib/stage-sidebar.ts | 7 + .../app/routes/run-stages-details.test.tsx | 87 ++++++++++- apps/fabro-web/app/routes/run-stages.tsx | 61 +++++++- docs/public/api-reference/fabro-api.yaml | 10 ++ lib/apps/fabro-server/src/demo/mod.rs | 1 + .../src/server/handler/billing.rs | 6 +- lib/apps/fabro-server/src/server/tests.rs | 141 ++++++++++++++++++ .../fabro-workflow/src/billing_rollup.rs | 29 +--- .../fabro-types/src/run_projection.rs | 86 ++++++++++- .../fabro-api-client/src/models/run-stage.ts | 7 + 11 files changed, 423 insertions(+), 34 deletions(-) diff --git a/apps/fabro-web/app/lib/stage-sidebar.test.ts b/apps/fabro-web/app/lib/stage-sidebar.test.ts index a71323654..e1cc9092b 100644 --- a/apps/fabro-web/app/lib/stage-sidebar.test.ts +++ b/apps/fabro-web/app/lib/stage-sidebar.test.ts @@ -17,6 +17,7 @@ function makeStage(nodeId: string, visit: number, status: StageState): Stage { duration: "--", startedAt: null, providerUsed: null, + billing: null, }; } @@ -38,6 +39,15 @@ describe("mapRunStagesToSidebarStages", () => { model: "gpt-5.5", reasoning_effort: "high", }, + billing: { + input_tokens: 28_640, + output_tokens: 7_550, + total_tokens: 43_690, + reasoning_tokens: 1_200, + cache_read_tokens: 4_800, + cache_write_tokens: 1_500, + total_usd_micros: 720_000, + }, }, { id: "apply-changes@2", @@ -46,6 +56,14 @@ describe("mapRunStagesToSidebarStages", () => { status: "running", node_id: "apply", visit: 2, + billing: { + input_tokens: 0, + output_tokens: 0, + total_tokens: 0, + reasoning_tokens: 0, + cache_read_tokens: 0, + cache_write_tokens: 0, + }, }, ], meta: { has_more: false }, @@ -64,6 +82,10 @@ describe("mapRunStagesToSidebarStages", () => { model: "gpt-5.5", reasoning_effort: "high", }); + // Each visit keeps its own tokens and cost, so the stage popover never + // shows a sibling visit's usage. + expect(result[0].billing?.total_usd_micros).toBe(720_000); + expect(result[1].billing?.total_usd_micros).toBeUndefined(); expect(formatStageLabel(result[0])).toBe("Apply Changes"); expect(result[1].id).toBe("apply-changes@2"); diff --git a/apps/fabro-web/app/lib/stage-sidebar.ts b/apps/fabro-web/app/lib/stage-sidebar.ts index c165adee0..191681459 100644 --- a/apps/fabro-web/app/lib/stage-sidebar.ts +++ b/apps/fabro-web/app/lib/stage-sidebar.ts @@ -1,5 +1,6 @@ import { StageState } from "@qltysh/fabro-api-client"; import type { + BilledTokenCounts, PaginatedRunStageList, StageHandler, StageModelUsage, @@ -27,6 +28,11 @@ export interface Stage { resumedFromStageId: string | null; startedAt: string | null; providerUsed: StageModelUsage | null; + /** + * Tokens and cost for this visit alone, priced the same way the Billing tab + * prices its per-node rows. All-zero counts mean the stage called no model. + */ + billing: BilledTokenCounts | null; } export const ACTIVE_STAGE_STATES: ReadonlySet = new Set([ @@ -102,6 +108,7 @@ export function mapRunStagesToSidebarStages( : "--", startedAt: stage.started_at ?? null, providerUsed: stage.provider_used ?? null, + billing: stage.billing ?? null, }); } return stages; diff --git a/apps/fabro-web/app/routes/run-stages-details.test.tsx b/apps/fabro-web/app/routes/run-stages-details.test.tsx index da18f7e67..2f453c63e 100644 --- a/apps/fabro-web/app/routes/run-stages-details.test.tsx +++ b/apps/fabro-web/app/routes/run-stages-details.test.tsx @@ -1,9 +1,13 @@ import { describe, expect, test } from "bun:test"; import { renderToStaticMarkup } from "react-dom/server"; -import type { ReasoningOutput } from "@qltysh/fabro-api-client"; +import type { + BilledTokenCounts, + ReasoningOutput, + StageModelUsage, +} from "@qltysh/fabro-api-client"; -import { EventDetails } from "./run-stages"; +import { EventDetails, ModelUsagePopover } from "./run-stages"; const RUN_START = "2026-04-09T12:00:00Z"; @@ -72,3 +76,82 @@ describe("EventDetails reasoning", () => { expect(html).toContain(`${"x".repeat(280)}…`); }); }); + +const PROVIDER_USED: StageModelUsage = { + mode: "agent", + provider: "moonshot", + model: "kimi-k3", + reasoning_effort: "max", +}; + +function billing(partial: Partial): BilledTokenCounts { + return { + input_tokens: 0, + output_tokens: 0, + total_tokens: 0, + reasoning_tokens: 0, + cache_read_tokens: 0, + cache_write_tokens: 0, + ...partial, + }; +} + +function popoverMarkup(counts: BilledTokenCounts | null): string { + return renderToStaticMarkup( + , + ); +} + +describe("ModelUsagePopover billing", () => { + test("shows the visit's token buckets and cost next to the model", () => { + const html = popoverMarkup( + billing({ + input_tokens: 28_640, + output_tokens: 7_550, + reasoning_tokens: 1_200, + cache_read_tokens: 4_800, + cache_write_tokens: 1_500, + total_tokens: 43_690, + total_usd_micros: 720_000, + }), + ); + + expect(html).toContain("kimi-k3"); + expect(html).toContain("Cache read"); + expect(html).toContain("4.8k"); + expect(html).toContain("Cache creation"); + expect(html).toContain("1.5k"); + expect(html).toContain("Uncached"); + expect(html).toContain("28.6k"); + // Output folds in reasoning tokens, matching the Billing tab. + expect(html).toContain("Output"); + expect(html).toContain("8.8k"); + expect(html).toContain("Cost"); + expect(html).toContain("$0.72"); + }); + + test("omits the token section for a stage that called no model", () => { + const html = popoverMarkup(billing({})); + + expect(html).toContain("kimi-k3"); + expect(html).not.toContain("Tokens"); + expect(html).not.toContain("Cost"); + }); + + test("still shows tokens when nothing priced the stage", () => { + const html = popoverMarkup( + billing({ input_tokens: 1_000, output_tokens: 500, total_tokens: 1_500 }), + ); + + expect(html).toContain("Uncached"); + expect(html).toContain("1.0k"); + expect(html).not.toContain("Cost"); + }); + + test("renders the model rows alone when the stage list carried no billing", () => { + const html = popoverMarkup(null); + + expect(html).toContain("kimi-k3"); + expect(html).not.toContain("Tokens"); + }); +}); diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index 43f4410f1..246f3b2ed 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -67,6 +67,7 @@ import { formatBytes, formatDurationMs, formatTokenCount, + formatUsdMicros, } from "../lib/format"; import { plural } from "../lib/plural"; import { @@ -93,6 +94,7 @@ import { type UnknownRecord, } from "../lib/unknown"; import type { + BilledTokenCounts, EventEnvelope, ReasoningOutput, StageHandler, @@ -866,10 +868,59 @@ export function formatStageModelUsageLabel( return effort ? `${model}[${effort}]` : model; } -function ModelUsagePopover({ +const POPOVER_NUMBER = "block text-right font-mono tabular-nums"; + +/** + * The disjoint token buckets behind a stage's usage, labelled and ordered to + * match the Billing tab's breakdown so the two views read the same. `Uncached` + * is input that missed the cache; `Output` folds in reasoning tokens. + */ +function stageTokenBuckets(billing: BilledTokenCounts) { + return [ + { label: "Cache read", value: billing.cache_read_tokens }, + { label: "Cache creation", value: billing.cache_write_tokens }, + { label: "Uncached", value: billing.input_tokens }, + { + label: "Output", + value: billing.output_tokens + billing.reasoning_tokens, + }, + ]; +} + +/** Tokens and cost for this stage visit alone. */ +function StageBillingRows({ billing }: { billing: BilledTokenCounts }) { + const buckets = stageTokenBuckets(billing); + if (buckets.every((bucket) => bucket.value === 0)) return null; + const cost = formatUsdMicros(billing.total_usd_micros); + return ( +
+ Tokens + + {buckets.map((bucket) => ( + + + {bucket.value === 0 + ? "0" + : formatTokenCount(bucket.value, { compactDecimal: true })} + + + ))} + {cost && ( + + {cost} + + )} + +
+ ); +} + +export function ModelUsagePopover({ providerUsed, + billing, }: { providerUsed: StageModelUsage; + billing: BilledTokenCounts | null; }) { return ( <> @@ -892,6 +943,7 @@ function ModelUsagePopover({ {providerUsed.speed} )} + {billing && } ); } @@ -1905,6 +1957,7 @@ function EventsToolbar({ filteredCount, totalCount, providerUsed, + billing, events, runId, stageId, @@ -1924,6 +1977,7 @@ function EventsToolbar({ filteredCount: number; totalCount: number; providerUsed: StageModelUsage | null; + billing: BilledTokenCounts | null; events: EventEnvelope[]; runId: string; stageId: string; @@ -2004,7 +2058,9 @@ function EventsToolbar({ className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${ showFilters ? "" : "ml-auto" }`} - content={} + content={ + + } >