From 1f6e53c949257ca39d4d3e234c8a14769791aeaf Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 14:47:46 -0400 Subject: [PATCH 01/16] Share one poison-tolerant lock helper from fabro-util Seven crates' files each carried the same four-line lock function that recovers a poisoned mutex. fabro_util::sync::lock is that function, once; the copies in fabro-petri and fabro-server are gone. fabro-template's helper panics on poison instead, a different policy, and is left as is. Co-Authored-By: Claude Fable 5.1 --- lib/apps/fabro-server/src/petri_runs.rs | 19 +++---- lib/components/fabro-petri/src/hooks.rs | 53 ++++++++++--------- lib/components/fabro-petri/src/http_store.rs | 15 +++--- lib/components/fabro-petri/src/projector.rs | 15 +++--- .../fabro-petri/src/projector/cache.rs | 13 ++--- lib/components/fabro-petri/src/run_store.rs | 17 +++--- .../fabro-petri/src/test_support.rs | 20 +++---- lib/foundation/fabro-util/src/lib.rs | 1 + lib/foundation/fabro-util/src/sync.rs | 11 ++++ 9 files changed, 81 insertions(+), 83 deletions(-) create mode 100644 lib/foundation/fabro-util/src/sync.rs diff --git a/lib/apps/fabro-server/src/petri_runs.rs b/lib/apps/fabro-server/src/petri_runs.rs index ba7efbdb5..16bef7e2b 100644 --- a/lib/apps/fabro-server/src/petri_runs.rs +++ b/lib/apps/fabro-server/src/petri_runs.rs @@ -14,12 +14,13 @@ //! `RunOptions::run_key`. use std::collections::HashMap; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use fabro_db::DbPool; use fabro_petri::SqliteRunStore; use fabro_petri::petri::{Access, OwnerId, RunKey, RunLogs, RunStore as _, StoreError}; use fabro_types::RunId; +use fabro_util::sync; use tracing::debug; pub(crate) struct PetriRuns { @@ -66,7 +67,7 @@ impl PetriRuns { ) -> Result, StoreError> { let handle = self.store.open(&Self::key(&run_id), access.clone()).await?; if let Some(owner) = access.owner() { - lock(&self.handles).insert((run_id, owner.clone()), Arc::clone(&handle)); + sync::lock(&self.handles).insert((run_id, owner.clone()), Arc::clone(&handle)); } Ok(handle) } @@ -80,7 +81,7 @@ impl PetriRuns { run_id: RunId, owner: &OwnerId, ) -> Result, StoreError> { - if let Some(handle) = lock(&self.handles).get(&(run_id, owner.clone())) { + if let Some(handle) = sync::lock(&self.handles).get(&(run_id, owner.clone())) { return Ok(Arc::clone(handle)); } let holder = self.store.owner(&Self::key(&run_id)).await?; @@ -107,7 +108,7 @@ impl PetriRuns { /// Drop the handle `owner` holds on the run: the worker's own release. /// The store ends the lease when this was the owner's last handle. pub(crate) fn release(&self, run_id: RunId, owner: &OwnerId) { - let handle = lock(&self.handles).remove(&(run_id, owner.clone())); + let handle = sync::lock(&self.handles).remove(&(run_id, owner.clone())); debug!( run_id = %run_id, owner = %owner, @@ -132,7 +133,7 @@ impl PetriRuns { /// releasing does not keep the lease. pub(crate) fn worker_exited(&self, run_id: RunId) { let dropped = { - let mut handles = lock(&self.handles); + let mut handles = sync::lock(&self.handles); let owners: Vec<_> = handles .keys() .filter(|(held, _)| *held == run_id) @@ -154,10 +155,6 @@ impl PetriRuns { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[cfg(test)] mod tests { use std::collections::BTreeMap; @@ -226,14 +223,14 @@ mod tests { } fn launched_mode(&self) -> Option<&'static str> { - *lock(&self.mode) + *sync::lock(&self.mode) } } #[async_trait::async_trait] impl WorkerRuntime for HeldWorkerRuntime { async fn start(&self, spec: WorkerLaunchSpec) -> anyhow::Result { - *lock(&self.mode) = Some(spec.mode); + *sync::lock(&self.mode) = Some(spec.mode); self.running.store(true, Ordering::SeqCst); let exit = Arc::clone(&self.exit); let stderr: Pin> = Box::pin(tokio::io::empty()); diff --git a/lib/components/fabro-petri/src/hooks.rs b/lib/components/fabro-petri/src/hooks.rs index e200f0a61..a9e503ff6 100644 --- a/lib/components/fabro-petri/src/hooks.rs +++ b/lib/components/fabro-petri/src/hooks.rs @@ -73,7 +73,7 @@ use std::collections::{BTreeMap, HashMap, HashSet}; use std::path::{Path, PathBuf}; -use std::sync::{Arc, Mutex, MutexGuard, OnceLock, PoisonError}; +use std::sync::{Arc, Mutex, OnceLock}; use std::time::Duration; use fabro_checkpoint::author::GitAuthor; @@ -86,6 +86,7 @@ use fabro_types::{ BlobHash, DiffSummary, GitIdentity, GitIdentitySource, RunId, SandboxProviderKind, }; use fabro_util::error::collect_chain; +use fabro_util::sync; use fabro_util::workspace_glob::{WorkspaceGlobError, WorkspaceGlobSet}; use petri_execution::{CancelReason, CoordinatorHandle, InvocationId, RunKey, RunStore}; use petri_runtime::driver::lifecycle::{ @@ -315,7 +316,7 @@ impl FabroHooks { /// engine reports the run failed with. #[must_use] pub fn checkpoint_failure(&self) -> Option { - lock(&self.failure).clone() + sync::lock(&self.failure).clone() } /// The run's workspaces on this host, as the hooks reach them. @@ -327,14 +328,14 @@ impl FabroHooks { /// The lock that serializes Git work in one workspace. fn workspace_lock(&self, workspace: &str) -> Arc> { Arc::clone( - lock(&self.workspace_locks) + sync::lock(&self.workspace_locks) .entry(workspace.to_string()) .or_default(), ) } fn fail_run(&self, message: &str) { - let mut failure = lock(&self.failure); + let mut failure = sync::lock(&self.failure); if failure.is_none() { *failure = Some(message.to_string()); } @@ -358,7 +359,9 @@ impl FabroHooks { if self.workspaces.workspace_exists(&isolated).await { return Ok(isolated); } - let cached = lock(&self.inherited).get(&context.invocation).cloned(); + let cached = sync::lock(&self.inherited) + .get(&context.invocation) + .cloned(); let inherited = if let Some(inherited) = cached { inherited } else { @@ -373,7 +376,7 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - lock(&self.inherited).insert(context.invocation, inherited.clone()); + sync::lock(&self.inherited).insert(context.invocation, inherited.clone()); inherited }; Ok(inherited.unwrap_or(isolated)) @@ -382,7 +385,9 @@ impl FabroHooks { /// The environment of `scope` in the context's execution, as /// `scope_acquired` kept it, with the workspace id the executor named. fn env_of(&self, context: &HookContext, scope: ScopeId) -> Option { - lock(&self.envs).get(&(context.execution, scope)).cloned() + sync::lock(&self.envs) + .get(&(context.execution, scope)) + .cloned() } /// The checkpoint commit for one attempt's result. `Ok(Some)` is the @@ -532,7 +537,7 @@ impl FabroHooks { /// Remember a commit this process made, and record the run branch when /// this commit created it. async fn committed(&self, key: CheckpointKey, workspace: &str, snapshot: &Snapshot) { - lock(&self.committed).insert(key, (workspace.to_string(), snapshot.sha.clone())); + sync::lock(&self.committed).insert(key, (workspace.to_string(), snapshot.sha.clone())); let Some(branched) = &snapshot.branched else { return; }; @@ -668,7 +673,7 @@ impl FabroHooks { /// seeded. async fn restore_host(&self, workspace: &str) -> Result<(), ScopeAcquiredError> { let targets = self.restore_targets().await?; - let target = lock(targets).remove(workspace); + let target = sync::lock(targets).remove(workspace); let Some(target) = target else { return Ok(()); }; @@ -700,7 +705,7 @@ impl FabroHooks { env: &Arc, ) -> Result<(), ScopeAcquiredError> { let targets = self.restore_targets().await?; - let target = lock(targets).remove(workspace); + let target = sync::lock(targets).remove(workspace); let Some(target) = target else { return Ok(()); }; @@ -735,10 +740,10 @@ impl FabroHooks { self.recorded_loaded .get_or_try_init(|| self.load_recorded()) .await?; - if lock(&self.recorded).contains(&key) { + if sync::lock(&self.recorded).contains(&key) { return Ok(()); } - let committed = lock(&self.committed).get(&key).cloned(); + let committed = sync::lock(&self.committed).get(&key).cloned(); let (workspace, sha) = if let Some(committed) = committed { committed } else { @@ -806,8 +811,8 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - lock(&self.recorded).insert(key); - *lock(&self.last_checkpoint) = Some((workspace, sha)); + sync::lock(&self.recorded).insert(key); + *sync::lock(&self.last_checkpoint) = Some((workspace, sha)); Ok(()) } @@ -868,7 +873,7 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - let mut recorded = lock(&self.recorded); + let mut recorded = sync::lock(&self.recorded); let mut last = None; for record in stored { let PlatformRecord::Checkpoint(checkpoint) = &record.record else { @@ -888,7 +893,7 @@ impl FabroHooks { } } drop(recorded); - let mut last_checkpoint = lock(&self.last_checkpoint); + let mut last_checkpoint = sync::lock(&self.last_checkpoint); if last_checkpoint.is_none() { *last_checkpoint = last; } @@ -941,7 +946,7 @@ impl FabroHooks { }; let digest = BlobHash::new(&bytes); let identity = (path.clone(), digest.to_string()); - if lock(&self.collected).contains(&identity) { + if sync::lock(&self.collected).contains(&identity) { continue; } let blob = blobs @@ -974,7 +979,7 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - lock(&self.collected).insert(identity); + sync::lock(&self.collected).insert(identity); total_bytes = total_bytes.saturating_add(size); collected += 1; } @@ -994,7 +999,7 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - let mut collected = lock(&self.collected); + let mut collected = sync::lock(&self.collected); for record in stored { if let PlatformRecord::ArtifactCollected(artifact) = record.record { collected.insert((artifact.path, artifact.digest)); @@ -1022,7 +1027,7 @@ impl FabroHooks { let Some(base_sha) = branch.base_sha.clone() else { return Ok(()); }; - let last = lock(&self.last_checkpoint).clone(); + let last = sync::lock(&self.last_checkpoint).clone(); let Some((workspace, head_sha)) = last else { debug!(run_id = %self.run_id, "no checkpoint is recorded; no run diff"); return Ok(()); @@ -1149,10 +1154,6 @@ fn select_artifacts(mut candidates: Vec<(String, u64)>) -> Vec<(String, u64)> { selected } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - fn is_checkpoint_failure(status: &Status) -> bool { matches!(status, Status::Failure(info) if info.class.as_str() == CHECKPOINT_FAILED_CLASS) } @@ -1297,7 +1298,7 @@ impl ExecutionHooks for FabroHooks { ); let scope = released.scope; let notes = self.inner.scope_released(context, released).await; - lock(&self.envs).remove(&(context.execution, scope)); + sync::lock(&self.envs).remove(&(context.execution, scope)); notes } @@ -1308,7 +1309,7 @@ impl ExecutionHooks for FabroHooks { ) -> Result<(), ScopeAcquiredError> { self.inner.scope_acquired(context, acquired.clone()).await?; let workspace = acquired.workspace.as_str().to_owned(); - lock(&self.envs).insert( + sync::lock(&self.envs).insert( (context.execution, acquired.scope), (workspace.clone(), Arc::clone(&acquired.env)), ); diff --git a/lib/components/fabro-petri/src/http_store.rs b/lib/components/fabro-petri/src/http_store.rs index 76ee9b69c..1a08588ee 100644 --- a/lib/components/fabro-petri/src/http_store.rs +++ b/lib/components/fabro-petri/src/http_store.rs @@ -62,13 +62,14 @@ use std::collections::HashMap; use std::future::Future; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError, Weak}; +use std::sync::{Arc, Mutex, Weak}; use std::time::Duration; use std::{fmt, mem, ptr}; use fabro_api::types::{PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriRecord}; use fabro_client::{Client, api_failure_for}; use fabro_types::{BlobHash, RunId}; +use fabro_util::sync; use petri_store::{Access, Digest, LogId, OwnerId, Record, RunKey, RunLogs, RunStore, StoreError}; use serde_json::Value; use tokio::runtime::Handle; @@ -148,7 +149,7 @@ impl HttpRunStore { owner: OwnerId, locator: String, ) -> Arc { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let slot = (key.clone(), owner.clone()); if let Some(handle) = live.get(&slot).and_then(Weak::upgrade) { return handle; @@ -185,7 +186,7 @@ impl Shared { /// Await every release a dropped handle spawned, so what follows sees /// the lease as the drops left it. async fn drain_releases(&self) { - let pending = mem::take(&mut *lock(&self.releases)); + let pending = mem::take(&mut *sync::lock(&self.releases)); for release in pending { // A release task never panics: it reports its own failure. let _ = release.await; @@ -436,7 +437,7 @@ impl Drop for HttpRunLogs { return; }; { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let slot = (self.key.clone(), owner.clone()); let this: *const Self = self; if live @@ -454,7 +455,7 @@ impl Drop for HttpRunLogs { let release = runtime.spawn(async move { shared.release(&key, run_id, &owner).await; }); - lock(&self.shared.releases).push(release); + sync::lock(&self.shared.releases).push(release); } Err(_) => { warn!( @@ -551,10 +552,6 @@ impl RunLogs for HttpRunLogs { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[cfg(test)] mod tests { use fabro_client::ApiFailure; diff --git a/lib/components/fabro-petri/src/projector.rs b/lib/components/fabro-petri/src/projector.rs index 2dea3dcf0..b9b3e274d 100644 --- a/lib/components/fabro-petri/src/projector.rs +++ b/lib/components/fabro-petri/src/projector.rs @@ -57,7 +57,7 @@ mod cache; use std::collections::{BTreeMap, BTreeSet, HashMap}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use std::time::Duration; use fabro_db::DbPool; @@ -65,6 +65,7 @@ use fabro_store::platform_records::{PlatformRecordStore, StoredPlatformRecord, n use fabro_store::{RunProjection, RunSummaryStore}; use fabro_types::{RunId, RunStreamItem, RunStreamItemKind}; use fabro_util::error::collect_chain; +use fabro_util::sync; use petri_execution::events::{self, EventId, EventSource, RunEvent}; use petri_execution::{Access, CoordinatorEvent, RunKey, RunStore as _, inspect}; use petri_runtime::engine::Event; @@ -270,7 +271,7 @@ impl Projector { .map_err(ProjectError::Database)?; } records.commit().await.map_err(ProjectError::Database)?; - lock(&self.slots).remove(&run_id); + sync::lock(&self.slots).remove(&run_id); Ok(()) } @@ -290,7 +291,7 @@ impl Projector { /// more when it ends; any number of signals in between coalesce. pub fn signal(self: &Arc, run_id: RunId) { { - let mut slots = lock(&self.slots); + let mut slots = sync::lock(&self.slots); let slot = slots.entry(run_id).or_default(); if slot.running { slot.pending = true; @@ -312,7 +313,7 @@ impl Projector { false } }; - let mut slots = lock(&projector.slots); + let mut slots = sync::lock(&projector.slots); let slot = slots.entry(run_id).or_default(); if again || slot.pending { slot.pending = false; @@ -329,7 +330,7 @@ impl Projector { pub async fn settle(&self, run_id: RunId) { loop { let idle = { - let slots = lock(&self.slots); + let slots = sync::lock(&self.slots); slots .get(&run_id) .is_none_or(|slot| !slot.running && !slot.pending) @@ -1059,10 +1060,6 @@ fn column(value: u64) -> i64 { i64::try_from(value).unwrap_or(i64::MAX) } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - /// The run's projection rebuilt from its records alone, with nothing /// stored: what a fresh projector would commit over the same records. A test /// compares it with the live view. `records` and `views` are the two pools diff --git a/lib/components/fabro-petri/src/projector/cache.rs b/lib/components/fabro-petri/src/projector/cache.rs index 3791b0ced..c2f423443 100644 --- a/lib/components/fabro-petri/src/projector/cache.rs +++ b/lib/components/fabro-petri/src/projector/cache.rs @@ -15,10 +15,11 @@ //! rows: the rebuild test in `tests/projection.rs` compares the two. use std::collections::HashMap; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; use fabro_types::RunId; +use fabro_util::sync; use petri_execution::events::{RunEvent, RunReplay}; use tokio::sync::Mutex as AsyncMutex; @@ -86,7 +87,7 @@ impl Caches { /// The run's pass lock, holding its cache if one is kept; the run counts /// as used now. pub(super) fn pass_of(&self, run_id: RunId) -> Arc>> { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let entry = runs.entry(run_id).or_insert_with(|| Entry { pass: Arc::default(), touched: Instant::now(), @@ -99,7 +100,7 @@ impl Caches { /// with no cache and no pass under way. A run whose pass is running is /// in use and left alone. How many caches were dropped. pub(crate) fn sweep(&self, idle: Duration) -> usize { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let mut dropped = 0; runs.retain(|_, entry| { if entry.touched.elapsed() < idle { @@ -121,12 +122,8 @@ impl Caches { /// Whether a cache is kept for the run: a test's view of the cache. pub(crate) fn holds(&self, run_id: RunId) -> bool { - let runs = lock(&self.runs); + let runs = sync::lock(&self.runs); runs.get(&run_id) .is_some_and(|entry| entry.pass.try_lock().is_ok_and(|cache| cache.is_some())) } } - -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} diff --git a/lib/components/fabro-petri/src/run_store.rs b/lib/components/fabro-petri/src/run_store.rs index d1eae1bdb..683e74f2f 100644 --- a/lib/components/fabro-petri/src/run_store.rs +++ b/lib/components/fabro-petri/src/run_store.rs @@ -55,13 +55,14 @@ use std::collections::HashMap; use std::error::Error; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError, Weak}; +use std::sync::{Arc, Mutex, Weak}; use std::time::{SystemTime, UNIX_EPOCH}; use std::{fmt, mem, ptr}; use fabro_db::DbPool; use fabro_store::BlobStore; use fabro_types::BlobHash; +use fabro_util::sync; use petri_store::{ Access, Digest, ExecutionId, LogId, OwnerId, Record, RunKey, RunLogs, RunStore, StoreError, }; @@ -150,7 +151,7 @@ impl SqliteRunStore { if result.rows_affected() == 0 { return Err(self.shared.not_found(key)); } - lock(&self.shared.live).remove(key); + sync::lock(&self.shared.live).remove(key); debug!(run_id = %key, "Petri run lease released from outside"); Ok(()) } @@ -174,7 +175,7 @@ impl SqliteRunStore { /// The writer handle for `owner`, once the lease is taken: the live one /// when this owner already holds a handle here, else a new one. fn writer(&self, key: &RunKey, owner: OwnerId) -> Arc { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); if let Some(handle) = live.get(key).and_then(Weak::upgrade) { if handle.owner.as_ref() == Some(&owner) { return handle; @@ -214,7 +215,7 @@ impl Shared { /// Await every release a dropped handle spawned, so what follows sees /// the lease as the drops left it. async fn drain_releases(&self) { - let pending = mem::take(&mut *lock(&self.releases)); + let pending = mem::take(&mut *sync::lock(&self.releases)); for release in pending { // A release task never panics: it reports its own failure. let _ = release.await; @@ -409,7 +410,7 @@ impl Drop for SqliteRunLogs { return; }; { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let this: *const Self = self; if live .get(&self.key) @@ -425,7 +426,7 @@ impl Drop for SqliteRunLogs { let release = runtime.spawn(async move { shared.release_owner(&key, &owner).await; }); - lock(&self.shared.releases).push(release); + sync::lock(&self.shared.releases).push(release); } Err(_) => { warn!( @@ -558,10 +559,6 @@ impl RunLogs for SqliteRunLogs { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - /// Milliseconds since the Unix epoch, as SQLite stores them. fn now_ms() -> i64 { SystemTime::now() diff --git a/lib/components/fabro-petri/src/test_support.rs b/lib/components/fabro-petri/src/test_support.rs index 7174cfbc5..38b37ce07 100644 --- a/lib/components/fabro-petri/src/test_support.rs +++ b/lib/components/fabro-petri/src/test_support.rs @@ -5,7 +5,7 @@ //! dev-dependency turns on. use std::collections::HashMap; -use std::sync::{Mutex, MutexGuard, PoisonError}; +use std::sync::Mutex; use std::time::Duration; use async_trait::async_trait; @@ -13,6 +13,7 @@ use bytes::Bytes; use fabro_store::platform_records::now_ms; use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord}; use fabro_types::{BlobHash, RunId}; +use fabro_util::sync; pub use petri_testkit::run_store; use crate::blobs::Blobs; @@ -47,7 +48,7 @@ impl MemoryBlobs { /// How many blobs the table holds. #[must_use] pub fn len(&self) -> usize { - lock(&self.rows).len() + sync::lock(&self.rows).len() } #[must_use] @@ -60,12 +61,12 @@ impl MemoryBlobs { impl Blobs for MemoryBlobs { async fn write(&self, bytes: &[u8]) -> anyhow::Result { let hash = BlobHash::new(bytes); - lock(&self.rows).insert(hash, bytes.to_vec()); + sync::lock(&self.rows).insert(hash, bytes.to_vec()); Ok(hash) } async fn read(&self, hash: &BlobHash) -> anyhow::Result> { - Ok(lock(&self.rows) + Ok(sync::lock(&self.rows) .get(hash) .map(|bytes| Bytes::copy_from_slice(bytes))) } @@ -86,14 +87,13 @@ impl MemoryPlatformRecords { /// Every record of the run, in seq order. #[must_use] pub fn records(&self, run_id: &RunId) -> Vec { - lock(&self.runs).get(run_id).cloned().unwrap_or_default() + sync::lock(&self.runs) + .get(run_id) + .cloned() + .unwrap_or_default() } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[async_trait] impl PlatformRecords for MemoryPlatformRecords { async fn append( @@ -102,7 +102,7 @@ impl PlatformRecords for MemoryPlatformRecords { record: &PlatformRecord, position: Option, ) -> Result { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let records = runs.entry(*run_id).or_default(); let stored = StoredPlatformRecord { seq: records.len() as u64 + 1, diff --git a/lib/foundation/fabro-util/src/lib.rs b/lib/foundation/fabro-util/src/lib.rs index c759f7171..28d4545e3 100644 --- a/lib/foundation/fabro-util/src/lib.rs +++ b/lib/foundation/fabro-util/src/lib.rs @@ -12,6 +12,7 @@ pub mod printer; pub mod run_log; pub mod session_secret; pub mod shell; +pub mod sync; pub mod terminal; pub mod text; pub mod time; diff --git a/lib/foundation/fabro-util/src/sync.rs b/lib/foundation/fabro-util/src/sync.rs new file mode 100644 index 000000000..851acdd4a --- /dev/null +++ b/lib/foundation/fabro-util/src/sync.rs @@ -0,0 +1,11 @@ +//! Locking helpers shared across crates. + +use std::sync::{Mutex, MutexGuard, PoisonError}; + +/// Lock a mutex, recovering the guard when another holder panicked. The +/// state such a mutex guards is bookkeeping (a cache, a set of ids, a +/// counter) that stays usable after a panic elsewhere, so the poison is +/// cleared rather than propagated. +pub fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex.lock().unwrap_or_else(PoisonError::into_inner) +} From 52b504d40b17e60413703a27ecb22617ad92d620 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 14:51:04 -0400 Subject: [PATCH 02/16] Take the workspace site as a parameter instead of paired host/sandbox methods RunWorkspaces already ran every git command through a private Site (a host path or a sandbox environment) but exposed each operation twice, as commit/commit_in, matches/matches_in, has_commit/has_commit_in, reset/reset_in, restore/restore_in and workspace_head/workspace_head_in. The pairs propagated into every caller: recovery had bring_host_to and bring_sandbox_to, the hooks had snapshot and snapshot_in_sandbox, and restore_host and restore_sandbox. Site is now the public parameter, each operation exists once, and the callers collapse to one function each. The hooks resolve a scope's workspace and site in one place, site_of. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/checkpoint.rs | 211 ++++++++----------- lib/components/fabro-petri/src/hooks.rs | 175 +++++---------- lib/components/fabro-petri/src/recovery.rs | 75 ++----- 3 files changed, 159 insertions(+), 302 deletions(-) diff --git a/lib/components/fabro-petri/src/checkpoint.rs b/lib/components/fabro-petri/src/checkpoint.rs index 48e360d1b..3564c208b 100644 --- a/lib/components/fabro-petri/src/checkpoint.rs +++ b/lib/components/fabro-petri/src/checkpoint.rs @@ -340,53 +340,19 @@ impl RunWorkspaces { .unwrap_or(false) } - /// The host site of a workspace. - fn host(&self, workspace: &str) -> Site { + /// The host site of a workspace: where `git` runs for a workspace kept + /// on this host. + #[must_use] + pub fn host(&self, workspace: &str) -> Site { Site::Host(self.workspace_path(workspace)) } /// Commit the workspace's files on the run branch as the snapshot of /// `key`, and publish it. An earlier commit of the same key that the - /// workspace still sits on, unchanged, is reused. + /// workspace still sits on, unchanged, is reused. On a sandbox site + /// `git` runs in the scope through its environment, and the commit + /// reaches the snapshot repository as a bundle. pub async fn commit( - &self, - workspace: &str, - key: CheckpointKey, - node: &str, - status: &str, - ) -> Result { - if !self.workspace_exists(workspace).await { - return Err(CheckpointError::WorkspaceMissing { - workspace: workspace.to_string(), - path: self.workspace_path(workspace), - }); - } - self.commit_at(&self.host(workspace), workspace, key, node, status) - .await - } - - /// [`commit`](Self::commit) for a workspace inside a sandbox: `git` - /// runs in the scope through `env`, and the commit reaches the - /// snapshot repository as a bundle. - pub async fn commit_in( - &self, - env: &Arc, - workspace: &str, - key: CheckpointKey, - node: &str, - status: &str, - ) -> Result { - self.commit_at( - &Site::Sandbox(Arc::clone(env)), - workspace, - key, - node, - status, - ) - .await - } - - async fn commit_at( &self, site: &Site, workspace: &str, @@ -394,6 +360,14 @@ impl RunWorkspaces { node: &str, status: &str, ) -> Result { + if let Site::Host(path) = site { + if !fs::try_exists(path).await.unwrap_or(false) { + return Err(CheckpointError::WorkspaceMissing { + workspace: workspace.to_string(), + path: path.clone(), + }); + } + } let branched = self.ensure_repository(site).await?; if let Some(existing) = self.published_sha(workspace, key).await? { if self.head(site).await?.as_deref() == Some(existing.as_str()) @@ -575,59 +549,20 @@ impl RunWorkspaces { .is_some()) } - /// The workspace's `HEAD`, or `None` when it has no commit. - pub async fn workspace_head(&self, workspace: &str) -> Result, CheckpointError> { - self.head(&self.host(workspace)).await - } - - /// [`workspace_head`](Self::workspace_head) for a workspace inside a - /// sandbox. - pub async fn workspace_head_in( - &self, - env: &Arc, - ) -> Result, CheckpointError> { - self.head(&Site::Sandbox(Arc::clone(env))).await - } - /// Whether the workspace sits on `sha` with nothing changed since. - pub async fn matches(&self, workspace: &str, sha: &str) -> Result { - self.matches_at(&self.host(workspace), sha).await - } - - /// [`matches`](Self::matches) for a workspace inside a sandbox. - pub async fn matches_in( - &self, - env: &Arc, - sha: &str, - ) -> Result { - self.matches_at(&Site::Sandbox(Arc::clone(env)), sha).await - } - - async fn matches_at(&self, site: &Site, sha: &str) -> Result { + pub async fn matches(&self, site: &Site, sha: &str) -> Result { Ok(self.head(site).await?.as_deref() == Some(sha) && self.is_clean(site).await?) } - /// Whether the host workspace's repository holds the commit `sha`, so a - /// reset can reach it; a directory that is no repository holds none. - pub async fn has_commit(&self, workspace: &str, sha: &str) -> Result { - if !self.workspace_exists(workspace).await { - return Ok(false); + /// Whether the workspace's repository holds the commit `sha`, so a + /// reset can reach it (without a transfer, in a sandbox); a directory + /// that is gone or is no repository holds none. + pub async fn has_commit(&self, site: &Site, sha: &str) -> Result { + if let Site::Host(path) = site { + if !fs::try_exists(path).await.unwrap_or(false) { + return Ok(false); + } } - self.has_commit_at(&self.host(workspace), sha).await - } - - /// Whether a sandbox workspace's repository holds the commit `sha`, so - /// a reset can reach it without a transfer. - pub async fn has_commit_in( - &self, - env: &Arc, - sha: &str, - ) -> Result { - self.has_commit_at(&Site::Sandbox(Arc::clone(env)), sha) - .await - } - - async fn has_commit_at(&self, site: &Site, sha: &str) -> Result { if self .git_status(site, "rev-parse", &["rev-parse", "--git-dir"]) .await? @@ -647,16 +582,7 @@ impl RunWorkspaces { /// Bring the workspace back to `sha`: tracked files reset, untracked /// files removed, the excluded caches left alone. - pub async fn reset(&self, workspace: &str, sha: &str) -> Result<(), CheckpointError> { - self.reset_at(&self.host(workspace), sha).await - } - - /// [`reset`](Self::reset) for a workspace inside a sandbox. - pub async fn reset_in(&self, env: &Arc, sha: &str) -> Result<(), CheckpointError> { - self.reset_at(&Site::Sandbox(Arc::clone(env)), sha).await - } - - async fn reset_at(&self, site: &Site, sha: &str) -> Result<(), CheckpointError> { + pub async fn reset(&self, site: &Site, sha: &str) -> Result<(), CheckpointError> { self.git(site, "reset", &["reset", "-q", "--hard", sha]) .await?; let mut clean = vec!["clean".to_string(), "-fdq".to_string()]; @@ -673,21 +599,38 @@ impl RunWorkspaces { } /// Recreate a gone workspace from the published snapshot `key`, at - /// `sha`, on the run branch. + /// `sha`, on the run branch. On a host site the workspace directory is + /// created and the snapshot fetched from the repository beside it. Into + /// a sandbox the snapshot enters the scope as a bundle of the + /// checkpoint's ref, and the workspace, fresh or stale, is fetched from + /// it and forced onto the run branch at `sha`. pub async fn restore( &self, + site: &Site, workspace: &str, key: CheckpointKey, sha: &str, ) -> Result<(), CheckpointError> { - let path = self.workspace_path(workspace); - fs::create_dir_all(&path) + match site { + Site::Host(path) => self.restore_on_host(path, workspace, key, sha).await, + Site::Sandbox(env) => self.restore_in_sandbox(env, workspace, key, sha).await, + } + } + + async fn restore_on_host( + &self, + path: &Path, + workspace: &str, + key: CheckpointKey, + sha: &str, + ) -> Result<(), CheckpointError> { + fs::create_dir_all(path) .await .map_err(|source| CheckpointError::Io { - path: path.clone(), + path: path.to_path_buf(), source, })?; - let site = Site::Host(path); + let site = Site::Host(path.to_path_buf()); self.git(&site, "init", &["init", "-q"]).await?; let repository = self.snapshot_repository(workspace); let repository = repository.to_string_lossy().into_owned(); @@ -710,10 +653,7 @@ impl RunWorkspaces { self.verify_restored(&site, sha).await } - /// [`restore`](Self::restore) into a sandbox: the snapshot enters the - /// scope as a bundle of the checkpoint's ref, and the workspace, fresh - /// or stale, is fetched from it and forced onto the run branch at `sha`. - pub async fn restore_in( + async fn restore_in_sandbox( &self, env: &Arc, workspace: &str, @@ -766,7 +706,7 @@ impl RunWorkspaces { "FETCH_HEAD", ]) .await?; - self.reset_at(&site, "HEAD").await?; + self.reset(&site, "HEAD").await?; self.verify_restored(&site, sha).await } .await; @@ -1083,7 +1023,8 @@ impl RunWorkspaces { .await } - async fn head(&self, site: &Site) -> Result, CheckpointError> { + /// The workspace's `HEAD`, or `None` when it has no commit. + pub async fn head(&self, site: &Site) -> Result, CheckpointError> { self.git_status(site, "rev-parse", &["rev-parse", "-q", "--verify", "HEAD"]) .await } @@ -1360,7 +1301,13 @@ mod tests { }; let first = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect("the commit"); assert!(!first.reused); @@ -1370,7 +1317,13 @@ mod tests { "the first commit created the run branch in a fresh repository" ); let again = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect("the second commit"); assert_eq!(again, Snapshot { @@ -1408,7 +1361,7 @@ mod tests { ); assert!( workspaces - .matches(workspace, &first.sha) + .matches(&workspaces.host(workspace), &first.sha) .await .expect("matches") ); @@ -1422,12 +1375,12 @@ mod tests { .expect("an untracked file"); assert!( !workspaces - .matches(workspace, &first.sha) + .matches(&workspaces.host(workspace), &first.sha) .await .expect("matches") ); workspaces - .reset(workspace, &first.sha) + .reset(&workspaces.host(workspace), &first.sha) .await .expect("the reset"); assert_eq!( @@ -1449,7 +1402,7 @@ mod tests { "the snapshot repository still knows the commit" ); workspaces - .restore(workspace, key, &first.sha) + .restore(&workspaces.host(workspace), workspace, key, &first.sha) .await .expect("the restore"); assert_eq!( @@ -1460,7 +1413,7 @@ mod tests { ); assert_eq!( workspaces - .workspace_head(workspace) + .head(&workspaces.host(workspace)) .await .expect("the head"), Some(first.sha) @@ -1483,7 +1436,13 @@ mod tests { attempt: 1, }; let error = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect_err("the commit fails"); assert!(matches!(error, CheckpointError::Command { .. }), "{error}"); @@ -1537,7 +1496,13 @@ mod tests { attempt: 1, }; let first = workspaces - .commit(workspace, first_key, "start", "success") + .commit( + &workspaces.host(workspace), + workspace, + first_key, + "start", + "success", + ) .await .expect("the first commit"); assert_eq!( @@ -1555,7 +1520,13 @@ mod tests { attempt: 1, }; let second = workspaces - .commit(workspace, second_key, "write", "success") + .commit( + &workspaces.host(workspace), + workspace, + second_key, + "write", + "success", + ) .await .expect("the second commit"); assert_eq!(second.branched, None); diff --git a/lib/components/fabro-petri/src/hooks.rs b/lib/components/fabro-petri/src/hooks.rs index a9e503ff6..73dc9d6da 100644 --- a/lib/components/fabro-petri/src/hooks.rs +++ b/lib/components/fabro-petri/src/hooks.rs @@ -103,7 +103,8 @@ use tracing::{debug, info, warn}; use crate::blobs::Blobs; use crate::checkpoint::{ - CHECKPOINT_FAILED_CLASS, CheckpointKey, EXCLUDE_DIRS, RunWorkspaces, Snapshot, WorkspaceDiff, + CHECKPOINT_FAILED_CLASS, CheckpointKey, EXCLUDE_DIRS, RunWorkspaces, Site, Snapshot, + WorkspaceDiff, }; use crate::platform_records::PlatformRecords; use crate::recovery::{self, Plan, RestoreTarget}; @@ -390,6 +391,28 @@ impl FabroHooks { .cloned() } + /// Where `scope`'s workspace is and what to call it: on this host, the + /// directory the records name (`None` when it does not exist yet); in + /// a sandbox, the environment kept at `scope_acquired` (`None` before + /// the scope was acquired). + async fn site_of( + &self, + context: &HookContext, + scope: ScopeId, + ) -> Result, String> { + if !self.host_workspaces { + return Ok(self + .env_of(context, scope) + .map(|(workspace, env)| (workspace, Site::Sandbox(env)))); + } + let workspace = self.workspace_of(context, scope).await?; + if !self.workspaces.workspace_exists(&workspace).await { + return Ok(None); + } + let site = self.workspaces.host(&workspace); + Ok(Some((workspace, site))) + } + /// The checkpoint commit for one attempt's result. `Ok(Some)` is the /// note to record, `Ok(None)` nothing to record, `Err` the fatal /// failure message. @@ -402,15 +425,9 @@ impl FabroHooks { status: &Status, origin: ResultOrigin, ) -> Result, String> { - if !self.host_workspaces { - return self - .snapshot_in_sandbox(context, scope, key, node, status, origin) - .await; - } - let workspace = self.workspace_of(context, scope).await?; - if !self.workspaces.workspace_exists(&workspace).await { + let Some((workspace, site)) = self.site_of(context, scope).await? else { // A skipped node or a driver-made outcome may precede the scope's - // environment; nothing of the stage's is on disk to snapshot. + // environment; nothing of the stage's exists to snapshot. if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) { return Ok(Some(Note::new( CHECKPOINT_NOTE, @@ -418,22 +435,21 @@ impl FabroHooks { "execution": key.execution, "firing": key.firing, "attempt": key.attempt, - "workspace": workspace, - "skipped": "the workspace does not exist yet", + "skipped": "the scope has no workspace yet", }), ))); } return Err(format!( - "the workspace `{workspace}` of scope {scope} does not exist at {}", - self.workspaces.workspace_path(&workspace).display() + "scope {scope} of execution {} has no workspace to snapshot", + context.execution )); - } + }; self.gate("commit", node).await; let serialized = self.workspace_lock(&workspace); let _held = serialized.lock().await; match self .workspaces - .commit(&workspace, key, node, status.tag()) + .commit(&site, &workspace, key, node, status.tag()) .await { Ok(snapshot) => { @@ -444,6 +460,7 @@ impl FabroHooks { firing = key.firing, attempt = key.attempt, reused = snapshot.reused, + site = ?site, "checkpoint committed" ); self.committed(key, &workspace, &snapshot).await; @@ -466,74 +483,6 @@ impl FabroHooks { } } - /// [`snapshot`](Self::snapshot) for a workspace inside the scope's - /// sandbox, through the environment kept at `scope_acquired`. - async fn snapshot_in_sandbox( - &self, - context: &HookContext, - scope: ScopeId, - key: CheckpointKey, - node: &str, - status: &Status, - origin: ResultOrigin, - ) -> Result, String> { - let Some((workspace, env)) = self.env_of(context, scope) else { - // A skipped node or a driver-made outcome may precede the scope's - // environment; nothing of the stage's exists to snapshot. - if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) { - return Ok(Some(Note::new( - CHECKPOINT_NOTE, - json!({ - "execution": key.execution, - "firing": key.firing, - "attempt": key.attempt, - "skipped": "the scope has no environment yet", - }), - ))); - } - return Err(format!( - "scope {scope} of execution {} has no sandbox environment to snapshot in", - context.execution - )); - }; - self.gate("commit", node).await; - let serialized = self.workspace_lock(&workspace); - let _held = serialized.lock().await; - match self - .workspaces - .commit_in(&env, &workspace, key, node, status.tag()) - .await - { - Ok(snapshot) => { - debug!( - run_id = %self.run_id, - node, - execution = key.execution, - firing = key.firing, - attempt = key.attempt, - reused = snapshot.reused, - "checkpoint committed in the sandbox" - ); - self.committed(key, &workspace, &snapshot).await; - Ok(Some(Note::new( - CHECKPOINT_NOTE, - json!({ - "execution": key.execution, - "firing": key.firing, - "attempt": key.attempt, - "workspace": workspace, - "git_commit_sha": snapshot.sha, - "reused": snapshot.reused, - }), - ))) - } - Err(error) => Err(format!( - "the checkpoint commit of `{node}` in the sandbox failed: {}", - collect_chain(&error).join(": ") - )), - } - } - /// Remember a commit this process made, and record the run branch when /// this commit created it. async fn committed(&self, key: CheckpointKey, workspace: &str, snapshot: &Snapshot) { @@ -666,12 +615,12 @@ impl FabroHooks { .await } - /// Bring a host workspace to the snapshot the resumed run's durable - /// state names, once, at its first acquisition. After a restart the - /// server already brought it there, so this verifies; a fork's fresh - /// workspace is restored here from the snapshot repository the fork - /// seeded. - async fn restore_host(&self, workspace: &str) -> Result<(), ScopeAcquiredError> { + /// Bring a workspace to the snapshot the resumed run's durable state + /// names, once, at its first acquisition. After a restart the server + /// already brought a host workspace there, so this verifies; a fork's + /// fresh workspace is restored here from the snapshot repository the + /// fork seeded; a sandbox workspace is only reachable here. + async fn restore(&self, workspace: &str, site: &Site) -> Result<(), ScopeAcquiredError> { let targets = self.restore_targets().await?; let target = sync::lock(targets).remove(workspace); let Some(target) = target else { @@ -679,11 +628,11 @@ impl FabroHooks { }; let serialized = self.workspace_lock(workspace); let _held = serialized.lock().await; - let action = recovery::bring_host_to(&self.workspaces, workspace, &target) + let action = recovery::bring_to(&self.workspaces, site, workspace, &target) .await .map_err(|error| { ScopeAcquiredError::new(format!( - "the host workspace `{workspace}` could not be brought to its snapshot: {}", + "the workspace `{workspace}` could not be brought to its snapshot: {}", collect_chain(&error).join(": ") )) })?; @@ -692,39 +641,8 @@ impl FabroHooks { workspace, sha = target.sha, action = ?action, - "host workspace brought to its durable snapshot" - ); - Ok(()) - } - - /// Bring a sandbox workspace to the snapshot the resumed run's durable - /// state names, once, at its first acquisition. - async fn restore_sandbox( - &self, - workspace: &str, - env: &Arc, - ) -> Result<(), ScopeAcquiredError> { - let targets = self.restore_targets().await?; - let target = sync::lock(targets).remove(workspace); - let Some(target) = target else { - return Ok(()); - }; - let serialized = self.workspace_lock(workspace); - let _held = serialized.lock().await; - let action = recovery::bring_sandbox_to(&self.workspaces, env, workspace, &target) - .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the sandbox workspace `{workspace}` could not be brought to its snapshot: {}", - collect_chain(&error).join(": ") - )) - })?; - info!( - run_id = %self.run_id, - workspace, - sha = target.sha, - action = ?action, - "sandbox workspace brought to its durable snapshot" + site = ?site, + "workspace brought to its durable snapshot" ); Ok(()) } @@ -1316,11 +1234,12 @@ impl ExecutionHooks for FabroHooks { if !self.resumed { return Ok(()); } - if self.host_workspaces { - self.restore_host(&workspace).await + let site = if self.host_workspaces { + self.workspaces.host(&workspace) } else { - self.restore_sandbox(&workspace, &acquired.env).await - } + Site::Sandbox(Arc::clone(&acquired.env)) + }; + self.restore(&workspace, &site).await } } diff --git a/lib/components/fabro-petri/src/recovery.rs b/lib/components/fabro-petri/src/recovery.rs index 2c068ed86..76d56d7ea 100644 --- a/lib/components/fabro-petri/src/recovery.rs +++ b/lib/components/fabro-petri/src/recovery.rs @@ -31,7 +31,7 @@ //! the worker's run reaches, so its target is deferred, and the worker's //! hooks read the same plan and apply it through the scope's environment //! at `scope_acquired`, before the first attempt runs there -//! ([`bring_sandbox_to`]). +//! ([`bring_to`]). use std::collections::BTreeMap; use std::path::PathBuf; @@ -45,11 +45,12 @@ use fabro_types::{RunId, SandboxProviderKind}; use petri_execution::host::{self, HostError}; use petri_execution::inspect::{self, ExecutionInspection, InspectError}; use petri_execution::{Access, InvocationId, RunKey, RunStore}; -use petri_runtime::executor::ExecEnv; use petri_store::StoreError; use tracing::info; -use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces}; +use crate::checkpoint::{ + CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces, Site, +}; use crate::platform_records::{PlatformRecordError, PlatformRecords}; use crate::workspace::{WorkspaceLookup, WorkspaceLookupError}; @@ -321,7 +322,13 @@ pub async fn recover(request: RecoveryRequest) -> Result Result { @@ -484,64 +493,22 @@ pub async fn bring_host_to( source, }; if workspaces - .has_commit(workspace, &target.sha) + .has_commit(site, &target.sha) .await .map_err(failed)? { if workspaces - .matches(workspace, &target.sha) + .matches(site, &target.sha) .await .map_err(failed)? { return Ok(WorkspaceAction::Verified); } - workspaces - .reset(workspace, &target.sha) - .await - .map_err(failed)?; + workspaces.reset(site, &target.sha).await.map_err(failed)?; return Ok(WorkspaceAction::Reset); } workspaces - .restore(workspace, target.key, &target.sha) - .await - .map_err(failed)?; - Ok(WorkspaceAction::Restored) -} - -/// Verify, reset or restore a sandbox workspace onto its target, through -/// the scope's environment: a retained sandbox that still holds the commit -/// is verified or reset in place; a fresh one, or one whose repository -/// lost the commit, is restored from a bundle of the snapshot. -pub async fn bring_sandbox_to( - workspaces: &RunWorkspaces, - env: &Arc, - workspace: &str, - target: &RestoreTarget, -) -> Result { - let failed = |source| RecoveryError::Workspace { - workspace: workspace.to_string(), - source, - }; - if workspaces - .has_commit_in(env, &target.sha) - .await - .map_err(failed)? - { - if workspaces - .matches_in(env, &target.sha) - .await - .map_err(failed)? - { - return Ok(WorkspaceAction::Verified); - } - workspaces - .reset_in(env, &target.sha) - .await - .map_err(failed)?; - return Ok(WorkspaceAction::Reset); - } - workspaces - .restore_in(env, workspace, target.key, &target.sha) + .restore(site, workspace, target.key, &target.sha) .await .map_err(failed)?; Ok(WorkspaceAction::Restored) From a039274c92027935d8892fffb2582f9d980d9673 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 14:53:46 -0400 Subject: [PATCH 03/16] Group the hooks' bookkeeping into three ledgers with one loaded-once shape FabroHooks held twenty-one flat fields. Three of them were "a set read from the store once, then kept current", in two shapes: a Mutex beside a OnceCell<()> that had to be initialised first, and, for the restore plan, a OnceCell> that could not be read uninitialised. The recorded checkpoints and the collected artifacts now use the second shape too. The checkpoint state (committed, recorded, last, branch), the artifact state (globs, collected) and the scope state (environments, inherited workspaces, per-workspace locks) are three structs with their own methods, so each invariant lives in one type instead of in every caller. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/hooks.rs | 360 ++++++++++++++---------- 1 file changed, 213 insertions(+), 147 deletions(-) diff --git a/lib/components/fabro-petri/src/hooks.rs b/lib/components/fabro-petri/src/hooks.rs index 73dc9d6da..04443fc71 100644 --- a/lib/components/fabro-petri/src/hooks.rs +++ b/lib/components/fabro-petri/src/hooks.rs @@ -194,61 +194,141 @@ type AcquiredEnv = (String, Arc); /// The identity of a collected file: its path and content digest. type ArtifactIdentity = (String, String); -/// The last checkpoint recorded: its workspace and commit. -type LastCheckpoint = (String, String); +/// A checkpoint's workspace and commit. +type WorkspaceCommit = (String, String); -/// Fabro's `ExecutionHooks`, around the hooks the runtime installed. -pub struct FabroHooks { - inner: Arc, - run_id: RunId, - records: Arc, - /// Where an artifact's bytes and a diff's patch go; `None` records - /// summaries alone. - blobs: Option>, - workspaces: RunWorkspaces, - lookup: WorkspaceLookup, - identity: GitIdentity, - artifact_globs: Result, - host_workspaces: bool, - test_gates: Option, - handle: OnceLock, +/// The run's checkpoints as the hooks track them: what this process +/// committed, what has its platform record, which was recorded last, and +/// the run branch. +#[derive(Default)] +struct CheckpointLedger { /// The workspace and commit of every checkpoint this process made. - committed: Mutex>, - /// Which checkpoints have their platform record, loaded from the store - /// once and kept up to date with every append. - recorded: Mutex>, - recorded_loaded: OnceCell<()>, + committed: Mutex>, + /// Which checkpoints have their platform record: read from the store + /// once, then kept current with every append. + recorded: OnceCell>>, /// The workspace and commit of the checkpoint recorded last: the head /// the run's diff is measured to. - last_checkpoint: Mutex>, + last: Mutex>, /// The run branch as recorded, once: read from the store, or written /// by the commit that created the branch. - branch: OnceCell, - /// Every artifact collected so far, by path and digest, loaded from the - /// store once and kept up to date with every append. - collected: Mutex>, - collected_loaded: OnceCell<()>, - /// Inherited workspaces resolved through the run's records. - inherited: Mutex>>, - /// One lock per workspace: the branches of a parallel node and a nested - /// invocation share their caller's workspace, and Git allows one index - /// operation at a time in it. - workspace_locks: Mutex>>>, - /// The checkpoint failure that ended the run, when one did. - failure: Mutex>, + branch: OnceCell, +} + +impl CheckpointLedger { + fn remember_commit(&self, key: CheckpointKey, workspace: &str, sha: &str) { + sync::lock(&self.committed).insert(key, (workspace.to_string(), sha.to_string())); + } + + fn commit_of(&self, key: CheckpointKey) -> Option { + sync::lock(&self.committed).get(&key).cloned() + } + + fn last(&self) -> Option { + sync::lock(&self.last).clone() + } + + fn set_last(&self, last: WorkspaceCommit) { + *sync::lock(&self.last) = Some(last); + } + + /// The last checkpoint the store named, kept only when this process + /// has not recorded one yet. + fn set_last_if_unset(&self, last: Option) { + let mut current = sync::lock(&self.last); + if current.is_none() { + *current = last; + } + } +} + +/// The run's artifacts as the hooks track them: which files are collected, +/// and which were collected already. +struct ArtifactLedger { + /// The `[run.artifacts] include` patterns, or why they do not parse. + globs: Result, + /// Every artifact collected so far, by path and digest: read from the + /// store once, then kept current with every append. + collected: OnceCell>>, +} + +/// Where each acquired scope's workspace is, and the locks that serialize +/// Git work in it. +#[derive(Default)] +struct ScopeEnvs { /// The environment of every acquired scope, by execution and scope, /// with the workspace id the executor named: where `git` runs when the /// workspaces are not on this host, and where artifacts are read from /// on every provider. Dropped at release. - envs: Mutex>, + acquired: Mutex>, + /// Inherited workspaces resolved through the run's records. + inherited: Mutex>>, + /// One lock per workspace: the branches of a parallel node and a nested + /// invocation share their caller's workspace, and Git allows one index + /// operation at a time in it. + locks: Mutex>>>, +} + +impl ScopeEnvs { + fn insert(&self, execution: ExecutionId, scope: ScopeId, env: AcquiredEnv) { + sync::lock(&self.acquired).insert((execution, scope), env); + } + + fn remove(&self, execution: ExecutionId, scope: ScopeId) { + sync::lock(&self.acquired).remove(&(execution, scope)); + } + + fn get(&self, execution: ExecutionId, scope: ScopeId) -> Option { + sync::lock(&self.acquired).get(&(execution, scope)).cloned() + } + + /// The inherited workspace of an invocation, once resolved: `None` + /// when not resolved yet, `Some(None)` when it inherits none. + fn inherited(&self, invocation: InvocationId) -> Option> { + sync::lock(&self.inherited).get(&invocation).cloned() + } + + fn remember_inherited(&self, invocation: InvocationId, inherited: Option) { + sync::lock(&self.inherited).insert(invocation, inherited); + } + + /// The lock that serializes Git work in one workspace. + fn lock_for(&self, workspace: &str) -> Arc> { + Arc::clone( + sync::lock(&self.locks) + .entry(workspace.to_string()) + .or_default(), + ) + } +} + +/// Fabro's `ExecutionHooks`, around the hooks the runtime installed. +pub struct FabroHooks { + inner: Arc, + run_id: RunId, + records: Arc, + /// Where an artifact's bytes and a diff's patch go; `None` records + /// summaries alone. + blobs: Option>, + workspaces: RunWorkspaces, + lookup: WorkspaceLookup, + identity: GitIdentity, + host_workspaces: bool, + test_gates: Option, + handle: OnceLock, + checkpoints: CheckpointLedger, + artifacts: ArtifactLedger, + scopes: ScopeEnvs, + /// The checkpoint failure that ended the run, when one did. + failure: Mutex>, /// Whether the run continues from its records: a sandbox workspace is /// then brought to its snapshot when its scope is first acquired. - resumed: bool, + resumed: bool, /// The snapshot every live sandbox workspace must sit on before work /// resumes in it, read once from the records; an entry leaves when it /// is applied. - restore: OnceCell>>, - store: Arc, + restore: OnceCell>>, + store: Arc, } impl FabroHooks { @@ -284,21 +364,16 @@ impl FabroHooks { workspaces, lookup: WorkspaceLookup::new(Arc::clone(&store), run_key), identity, - artifact_globs: WorkspaceGlobSet::try_new(&spec.artifacts), host_workspaces: spec.host_workspaces, test_gates: spec.test_gates, handle: OnceLock::new(), - committed: Mutex::default(), - recorded: Mutex::default(), - recorded_loaded: OnceCell::new(), - last_checkpoint: Mutex::default(), - branch: OnceCell::new(), - collected: Mutex::default(), - collected_loaded: OnceCell::new(), - inherited: Mutex::default(), - workspace_locks: Mutex::default(), + checkpoints: CheckpointLedger::default(), + artifacts: ArtifactLedger { + globs: WorkspaceGlobSet::try_new(&spec.artifacts), + collected: OnceCell::new(), + }, + scopes: ScopeEnvs::default(), failure: Mutex::default(), - envs: Mutex::default(), resumed, restore: OnceCell::new(), store, @@ -326,15 +401,6 @@ impl FabroHooks { &self.workspaces } - /// The lock that serializes Git work in one workspace. - fn workspace_lock(&self, workspace: &str) -> Arc> { - Arc::clone( - sync::lock(&self.workspace_locks) - .entry(workspace.to_string()) - .or_default(), - ) - } - fn fail_run(&self, message: &str) { let mut failure = sync::lock(&self.failure); if failure.is_none() { @@ -360,10 +426,7 @@ impl FabroHooks { if self.workspaces.workspace_exists(&isolated).await { return Ok(isolated); } - let cached = sync::lock(&self.inherited) - .get(&context.invocation) - .cloned(); - let inherited = if let Some(inherited) = cached { + let inherited = if let Some(inherited) = self.scopes.inherited(context.invocation) { inherited } else { let inherited = self @@ -377,7 +440,8 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - sync::lock(&self.inherited).insert(context.invocation, inherited.clone()); + self.scopes + .remember_inherited(context.invocation, inherited.clone()); inherited }; Ok(inherited.unwrap_or(isolated)) @@ -386,9 +450,7 @@ impl FabroHooks { /// The environment of `scope` in the context's execution, as /// `scope_acquired` kept it, with the workspace id the executor named. fn env_of(&self, context: &HookContext, scope: ScopeId) -> Option { - sync::lock(&self.envs) - .get(&(context.execution, scope)) - .cloned() + self.scopes.get(context.execution, scope) } /// Where `scope`'s workspace is and what to call it: on this host, the @@ -445,7 +507,7 @@ impl FabroHooks { )); }; self.gate("commit", node).await; - let serialized = self.workspace_lock(&workspace); + let serialized = self.scopes.lock_for(&workspace); let _held = serialized.lock().await; match self .workspaces @@ -486,7 +548,8 @@ impl FabroHooks { /// Remember a commit this process made, and record the run branch when /// this commit created it. async fn committed(&self, key: CheckpointKey, workspace: &str, snapshot: &Snapshot) { - sync::lock(&self.committed).insert(key, (workspace.to_string(), snapshot.sha.clone())); + self.checkpoints + .remember_commit(key, workspace, &snapshot.sha); let Some(branched) = &snapshot.branched else { return; }; @@ -520,6 +583,7 @@ impl FabroHooks { firing: key.firing, }; let branch = self + .checkpoints .branch .get_or_try_init(|| async { if let Some(stored) = self.stored_branch().await? { @@ -626,7 +690,7 @@ impl FabroHooks { let Some(target) = target else { return Ok(()); }; - let serialized = self.workspace_lock(workspace); + let serialized = self.scopes.lock_for(workspace); let _held = serialized.lock().await; let action = recovery::bring_to(&self.workspaces, site, workspace, &target) .await @@ -655,14 +719,11 @@ impl FabroHooks { scope: ScopeId, key: CheckpointKey, ) -> Result<(), String> { - self.recorded_loaded - .get_or_try_init(|| self.load_recorded()) - .await?; - if sync::lock(&self.recorded).contains(&key) { + let recorded = self.recorded_checkpoints().await?; + if sync::lock(recorded).contains(&key) { return Ok(()); } - let committed = sync::lock(&self.committed).get(&key).cloned(); - let (workspace, sha) = if let Some(committed) = committed { + let (workspace, sha) = if let Some(committed) = self.checkpoints.commit_of(key) { committed } else { let acquired = self.env_of(context, scope).map(|(workspace, _)| workspace); @@ -670,7 +731,7 @@ impl FabroHooks { Some(workspace) => workspace, None => self.workspace_of(context, scope).await?, }; - let serialized = self.workspace_lock(&workspace); + let serialized = self.scopes.lock_for(&workspace); let held = serialized.lock().await; let found = self.workspaces.find(&workspace, key).await; drop(held); @@ -729,8 +790,8 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - sync::lock(&self.recorded).insert(key); - *sync::lock(&self.last_checkpoint) = Some((workspace, sha)); + sync::lock(recorded).insert(key); + self.checkpoints.set_last((workspace, sha)); Ok(()) } @@ -780,42 +841,43 @@ impl FabroHooks { /// resume's reissued routing decisions must not record again, and /// where the run's diff is measured to when this process made no /// checkpoint yet. - async fn load_recorded(&self) -> Result<(), String> { - let stored = self - .records - .read_kind(&self.run_id, PlatformRecordKind::Checkpoint) + async fn recorded_checkpoints(&self) -> Result<&Mutex>, String> { + self.checkpoints + .recorded + .get_or_try_init(|| async { + let stored = self + .records + .read_kind(&self.run_id, PlatformRecordKind::Checkpoint) + .await + .map_err(|error| { + format!( + "the run's checkpoint records could not be read: {}", + collect_chain(&error).join(": ") + ) + })?; + let mut recorded = HashSet::new(); + let mut last = None; + for record in stored { + let PlatformRecord::Checkpoint(checkpoint) = &record.record else { + continue; + }; + if let Some(key) = checkpoint + .operation + .as_ref() + .and_then(CheckpointKey::from_operation) + { + recorded.insert(key); + } + if let (Some(workspace), Some(sha)) = + (&checkpoint.workspace, &checkpoint.git_commit_sha) + { + last = Some((workspace.clone(), sha.clone())); + } + } + self.checkpoints.set_last_if_unset(last); + Ok(Mutex::new(recorded)) + }) .await - .map_err(|error| { - format!( - "the run's checkpoint records could not be read: {}", - collect_chain(&error).join(": ") - ) - })?; - let mut recorded = sync::lock(&self.recorded); - let mut last = None; - for record in stored { - let PlatformRecord::Checkpoint(checkpoint) = &record.record else { - continue; - }; - if let Some(key) = checkpoint - .operation - .as_ref() - .and_then(CheckpointKey::from_operation) - { - recorded.insert(key); - } - if let (Some(workspace), Some(sha)) = - (&checkpoint.workspace, &checkpoint.git_commit_sha) - { - last = Some((workspace.clone(), sha.clone())); - } - } - drop(recorded); - let mut last_checkpoint = sync::lock(&self.last_checkpoint); - if last_checkpoint.is_none() { - *last_checkpoint = last; - } - Ok(()) } /// The artifacts of a finished attempt: every file of its workspace @@ -828,7 +890,7 @@ impl FabroHooks { scope: ScopeId, key: CheckpointKey, ) -> Result { - let globs = match &self.artifact_globs { + let globs = match &self.artifacts.globs { Ok(globs) => globs, Err(error) => return Err(format!("invalid run.artifacts.include pattern: {error}")), }; @@ -843,9 +905,7 @@ impl FabroHooks { let Some(blobs) = &self.blobs else { return Err("the run has no blob table to collect artifacts into".to_string()); }; - self.collected_loaded - .get_or_try_init(|| self.load_collected()) - .await?; + let already = self.collected_artifacts().await?; let candidates = list_artifacts(env.as_ref(), globs).await?; let limit = usize::try_from(ARTIFACT_MAX_FILE_BYTES).unwrap_or(usize::MAX); let mut collected = 0; @@ -864,7 +924,7 @@ impl FabroHooks { }; let digest = BlobHash::new(&bytes); let identity = (path.clone(), digest.to_string()); - if sync::lock(&self.collected).contains(&identity) { + if sync::lock(already).contains(&identity) { continue; } let blob = blobs @@ -897,7 +957,7 @@ impl FabroHooks { collect_chain(&error).join(": ") ) })?; - sync::lock(&self.collected).insert(identity); + sync::lock(already).insert(identity); total_bytes = total_bytes.saturating_add(size); collected += 1; } @@ -906,24 +966,32 @@ impl FabroHooks { /// The artifacts already collected for the run, read once: a file that /// is unchanged since it was collected is not collected again. - async fn load_collected(&self) -> Result<(), String> { - let stored = self - .records - .read_kind(&self.run_id, PlatformRecordKind::ArtifactCollected) + async fn collected_artifacts(&self) -> Result<&Mutex>, String> { + self.artifacts + .collected + .get_or_try_init(|| async { + let stored = self + .records + .read_kind(&self.run_id, PlatformRecordKind::ArtifactCollected) + .await + .map_err(|error| { + format!( + "the run's artifact records could not be read: {}", + collect_chain(&error).join(": ") + ) + })?; + let collected = stored + .into_iter() + .filter_map(|record| match record.record { + PlatformRecord::ArtifactCollected(artifact) => { + Some((artifact.path, artifact.digest)) + } + _ => None, + }) + .collect(); + Ok(Mutex::new(collected)) + }) .await - .map_err(|error| { - format!( - "the run's artifact records could not be read: {}", - collect_chain(&error).join(": ") - ) - })?; - let mut collected = sync::lock(&self.collected); - for record in stored { - if let PlatformRecord::ArtifactCollected(artifact) = record.record { - collected.insert((artifact.path, artifact.digest)); - } - } - Ok(()) } /// The run's diff: the run branch's last checkpoint against the base @@ -931,10 +999,8 @@ impl FabroHooks { /// Nothing is recorded for a run that never created its branch or /// never checkpointed. async fn record_run_diff(&self) -> Result<(), String> { - self.recorded_loaded - .get_or_try_init(|| self.load_recorded()) - .await?; - let branch = match self.branch.get() { + self.recorded_checkpoints().await?; + let branch = match self.checkpoints.branch.get() { Some(branch) => Some(branch.clone()), None => self.stored_branch().await?, }; @@ -945,8 +1011,7 @@ impl FabroHooks { let Some(base_sha) = branch.base_sha.clone() else { return Ok(()); }; - let last = sync::lock(&self.last_checkpoint).clone(); - let Some((workspace, head_sha)) = last else { + let Some((workspace, head_sha)) = self.checkpoints.last() else { debug!(run_id = %self.run_id, "no checkpoint is recorded; no run diff"); return Ok(()); }; @@ -1216,7 +1281,7 @@ impl ExecutionHooks for FabroHooks { ); let scope = released.scope; let notes = self.inner.scope_released(context, released).await; - sync::lock(&self.envs).remove(&(context.execution, scope)); + self.scopes.remove(context.execution, scope); notes } @@ -1227,8 +1292,9 @@ impl ExecutionHooks for FabroHooks { ) -> Result<(), ScopeAcquiredError> { self.inner.scope_acquired(context, acquired.clone()).await?; let workspace = acquired.workspace.as_str().to_owned(); - sync::lock(&self.envs).insert( - (context.execution, acquired.scope), + self.scopes.insert( + context.execution, + acquired.scope, (workspace.clone(), Arc::clone(&acquired.env)), ); if !self.resumed { From 50e64c729a3775be9ae7d3b16b61a829c42e209e Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 14:57:17 -0400 Subject: [PATCH 04/16] Carry the hooks' failures as a typed HookError until the Petri boundary Eleven private hook methods returned Result<_, String> and rendered their causes with the same collect_chain(...).join(": ") in fourteen places. The error-handling strategy reserves String for rendered projections. HookError keeps every failure with its source; it is rendered once, by HookError::render, at the four Petri boundaries that carry text: the adjusted outcome of a failed checkpoint, a transition's problems, a scope acquisition's error, and the log. CheckpointKey gains a Display so three messages stop spelling it out by hand. The rendered text is the same as before, which the failed-checkpoint tests assert on. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/checkpoint.rs | 10 + lib/components/fabro-petri/src/hooks.rs | 310 +++++++++++-------- 2 files changed, 194 insertions(+), 126 deletions(-) diff --git a/lib/components/fabro-petri/src/checkpoint.rs b/lib/components/fabro-petri/src/checkpoint.rs index 3564c208b..df82a04e6 100644 --- a/lib/components/fabro-petri/src/checkpoint.rs +++ b/lib/components/fabro-petri/src/checkpoint.rs @@ -178,6 +178,16 @@ impl CheckpointKey { } } +impl std::fmt::Display for CheckpointKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "execution {} firing {} attempt {}", + self.execution, self.firing, self.attempt + ) + } +} + /// Why a snapshot could not be taken, found or restored. #[derive(Debug, thiserror::Error)] pub enum CheckpointError { diff --git a/lib/components/fabro-petri/src/hooks.rs b/lib/components/fabro-petri/src/hooks.rs index 04443fc71..36914d942 100644 --- a/lib/components/fabro-petri/src/hooks.rs +++ b/lib/components/fabro-petri/src/hooks.rs @@ -94,7 +94,7 @@ use petri_runtime::driver::lifecycle::{ Prepared, Recorded, ResultOrigin, RunFinished, ScopeAcquired, ScopeAcquiredError, ScopeReleased, Transition, TransitionError, TransitionReport, }; -use petri_runtime::executor::ExecEnv; +use petri_runtime::executor::{EnvError, ExecEnv}; use petri_runtime::ir::{ExecutionId, FailureInfo, ScopeId, Status}; use serde_json::json; use tokio::sync::{Mutex as AsyncMutex, OnceCell}; @@ -103,12 +103,12 @@ use tracing::{debug, info, warn}; use crate::blobs::Blobs; use crate::checkpoint::{ - CHECKPOINT_FAILED_CLASS, CheckpointKey, EXCLUDE_DIRS, RunWorkspaces, Site, Snapshot, - WorkspaceDiff, + CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, EXCLUDE_DIRS, RunWorkspaces, Site, + Snapshot, WorkspaceDiff, }; -use crate::platform_records::PlatformRecords; -use crate::recovery::{self, Plan, RestoreTarget}; -use crate::workspace::{self, WorkspaceLookup}; +use crate::platform_records::{PlatformRecordError, PlatformRecords}; +use crate::recovery::{self, Plan, RecoveryError, RestoreTarget}; +use crate::workspace::{self, WorkspaceLookup, WorkspaceLookupError}; /// The note kind the hooks record on a firing about its checkpoint. pub const CHECKPOINT_NOTE: &str = "fabro.checkpoint"; @@ -130,6 +130,88 @@ const ARTIFACT_MAX_TOTAL_BYTES: u64 = 50 * 1024 * 1024; /// How deep a traversal root is listed. const ARTIFACT_LIST_DEPTH: usize = 64; +/// Why the hooks' own work at a point failed. Each variant keeps its +/// source, and the chain is rendered once, by [`HookError::render`], at the +/// Petri boundaries that carry text: the adjusted outcome of a failed +/// checkpoint, a transition's problems, a scope acquisition's error, and +/// the log. +#[derive(Debug, thiserror::Error)] +pub enum HookError { + #[error("the workspace of scope {scope} in invocation {invocation} could not be found")] + Lookup { + scope: ScopeId, + invocation: InvocationId, + #[source] + source: WorkspaceLookupError, + }, + #[error("scope {scope} of execution {execution} has no workspace to snapshot")] + NoWorkspace { + scope: ScopeId, + execution: ExecutionId, + }, + #[error("the checkpoint commit of `{node}` failed")] + Commit { + node: String, + #[source] + source: CheckpointError, + }, + #[error("the checkpoint commit could not be looked up")] + Find(#[source] CheckpointError), + #[error("no checkpoint commit exists for {key}")] + NoCommit { key: CheckpointKey }, + #[error("the {what} could not be computed")] + Diff { + /// `stage's diff` or `run's diff`. + what: &'static str, + #[source] + source: CheckpointError, + }, + #[error("the {what} could not be stored")] + Blob { + /// `patch`, or `artifact \`\``. + what: String, + #[source] + source: anyhow::Error, + }, + #[error("the {kind} record could not be written")] + Write { + kind: &'static str, + #[source] + source: PlatformRecordError, + }, + #[error("the run's {kind} records could not be read")] + Read { + kind: &'static str, + #[source] + source: PlatformRecordError, + }, + #[error("invalid run.artifacts.include pattern")] + Globs(#[source] Arc), + #[error("the run has no blob table to collect artifacts into")] + NoBlobs, + #[error("the workspace could not be listed below `{root}`")] + List { + root: String, + #[source] + source: EnvError, + }, + #[error("the run's restore plan could not be read")] + Plan(#[source] RecoveryError), + #[error("{reason}")] + Unresumable { reason: String }, + #[error(transparent)] + Restore(RecoveryError), +} + +impl HookError { + /// The error and its causes on one line, for the boundaries that carry + /// text. + #[must_use] + pub fn render(&self) -> String { + collect_chain(self).join(": ") + } +} + /// What Fabro's hooks need beside the run: where the platform records go, /// who authors the commits, the checkpoint settings, and which files are /// the run's artifacts. @@ -246,7 +328,7 @@ impl CheckpointLedger { /// and which were collected already. struct ArtifactLedger { /// The `[run.artifacts] include` patterns, or why they do not parse. - globs: Result, + globs: Result>, /// Every artifact collected so far, by path and digest: read from the /// store once, then kept current with every append. collected: OnceCell>>, @@ -369,7 +451,7 @@ impl FabroHooks { handle: OnceLock::new(), checkpoints: CheckpointLedger::default(), artifacts: ArtifactLedger { - globs: WorkspaceGlobSet::try_new(&spec.artifacts), + globs: WorkspaceGlobSet::try_new(&spec.artifacts).map_err(Arc::new), collected: OnceCell::new(), }, scopes: ScopeEnvs::default(), @@ -421,7 +503,11 @@ impl FabroHooks { /// The workspace id of `scope` in the context's invocation: the /// isolated name when its workspace exists, else the inherited one the /// records name, else the isolated name for the caller to report. - async fn workspace_of(&self, context: &HookContext, scope: ScopeId) -> Result { + async fn workspace_of( + &self, + context: &HookContext, + scope: ScopeId, + ) -> Result { let isolated = workspace::isolated_workspace(context.invocation, scope); if self.workspaces.workspace_exists(&isolated).await { return Ok(isolated); @@ -433,12 +519,10 @@ impl FabroHooks { .lookup .inherited(context.invocation) .await - .map_err(|error| { - format!( - "the workspace of scope {scope} in invocation {} could not be found: {}", - context.invocation, - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Lookup { + scope, + invocation: context.invocation, + source, })?; self.scopes .remember_inherited(context.invocation, inherited.clone()); @@ -461,7 +545,7 @@ impl FabroHooks { &self, context: &HookContext, scope: ScopeId, - ) -> Result, String> { + ) -> Result, HookError> { if !self.host_workspaces { return Ok(self .env_of(context, scope) @@ -477,7 +561,7 @@ impl FabroHooks { /// The checkpoint commit for one attempt's result. `Ok(Some)` is the /// note to record, `Ok(None)` nothing to record, `Err` the fatal - /// failure message. + /// failure. async fn snapshot( &self, context: &HookContext, @@ -486,7 +570,7 @@ impl FabroHooks { node: &str, status: &Status, origin: ResultOrigin, - ) -> Result, String> { + ) -> Result, HookError> { let Some((workspace, site)) = self.site_of(context, scope).await? else { // A skipped node or a driver-made outcome may precede the scope's // environment; nothing of the stage's exists to snapshot. @@ -501,10 +585,10 @@ impl FabroHooks { }), ))); } - return Err(format!( - "scope {scope} of execution {} has no workspace to snapshot", - context.execution - )); + return Err(HookError::NoWorkspace { + scope, + execution: context.execution, + }); }; self.gate("commit", node).await; let serialized = self.scopes.lock_for(&workspace); @@ -538,10 +622,10 @@ impl FabroHooks { }), ))) } - Err(error) => Err(format!( - "the checkpoint commit of `{node}` failed: {}", - collect_chain(&error).join(": ") - )), + Err(source) => Err(HookError::Commit { + node: node.to_string(), + source, + }), } } @@ -560,7 +644,7 @@ impl FabroHooks { .clone() .unwrap_or_else(|| snapshot.sha.clone()); if let Err(error) = self.record_branch(key, workspace, base_sha).await { - warn!(run_id = %self.run_id, error = %error, "the run branch was not recorded"); + warn!(run_id = %self.run_id, error = %error.render(), "the run branch was not recorded"); } } @@ -577,7 +661,7 @@ impl FabroHooks { key: CheckpointKey, workspace: &str, base_sha: String, - ) -> Result<(), String> { + ) -> Result<(), HookError> { let position = StagePosition { execution: key.execution, firing: key.firing, @@ -587,7 +671,7 @@ impl FabroHooks { .branch .get_or_try_init(|| async { if let Some(stored) = self.stored_branch().await? { - return Ok::<_, String>(stored); + return Ok::<_, HookError>(stored); } let record = RunBranchRecord { run_branch: Some(self.workspaces.run_branch()), @@ -601,11 +685,9 @@ impl FabroHooks { Some(position), ) .await - .map_err(|error| { - format!( - "the run branch record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "run branch", + source, })?; let identity = PlatformRecord::GitIdentity(GitIdentityRecord { identity: self.identity.clone(), @@ -613,11 +695,9 @@ impl FabroHooks { self.records .append(&self.run_id, &identity, Some(position)) .await - .map_err(|error| { - format!( - "the git identity record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "git identity", + source, })?; info!( run_id = %self.run_id, @@ -633,16 +713,14 @@ impl FabroHooks { } /// The run branch the store already holds, when a record exists. - async fn stored_branch(&self) -> Result, String> { + async fn stored_branch(&self) -> Result, HookError> { let stored = self .records .read_kind(&self.run_id, PlatformRecordKind::RunBranch) .await - .map_err(|error| { - format!( - "the run's branch record could not be read: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Read { + kind: "branch", + source, })?; Ok(stored.into_iter().find_map(|stored| match stored.record { PlatformRecord::RunBranch(record) => Some(record), @@ -652,9 +730,7 @@ impl FabroHooks { /// The restore plan of a resumed run, read once: what every live /// sandbox workspace must be brought to at its first acquisition. - async fn restore_targets( - &self, - ) -> Result<&Mutex>, ScopeAcquiredError> { + async fn restore_targets(&self) -> Result<&Mutex>, HookError> { self.restore .get_or_try_init(|| async { let plan = recovery::plan( @@ -664,16 +740,11 @@ impl FabroHooks { &self.workspaces, ) .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the run's restore plan could not be read: {}", - collect_chain(&error).join(": ") - )) - })?; + .map_err(HookError::Plan)?; match plan { Plan::Resume { targets } => Ok(Mutex::new(targets)), Plan::Start => Ok(Mutex::default()), - Plan::Failed { reason } => Err(ScopeAcquiredError::new(reason)), + Plan::Failed { reason } => Err(HookError::Unresumable { reason }), } }) .await @@ -684,7 +755,7 @@ impl FabroHooks { /// already brought a host workspace there, so this verifies; a fork's /// fresh workspace is restored here from the snapshot repository the /// fork seeded; a sandbox workspace is only reachable here. - async fn restore(&self, workspace: &str, site: &Site) -> Result<(), ScopeAcquiredError> { + async fn restore(&self, workspace: &str, site: &Site) -> Result<(), HookError> { let targets = self.restore_targets().await?; let target = sync::lock(targets).remove(workspace); let Some(target) = target else { @@ -694,12 +765,7 @@ impl FabroHooks { let _held = serialized.lock().await; let action = recovery::bring_to(&self.workspaces, site, workspace, &target) .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the workspace `{workspace}` could not be brought to its snapshot: {}", - collect_chain(&error).join(": ") - )) - })?; + .map_err(HookError::Restore)?; info!( run_id = %self.run_id, workspace, @@ -718,7 +784,7 @@ impl FabroHooks { context: &HookContext, scope: ScopeId, key: CheckpointKey, - ) -> Result<(), String> { + ) -> Result<(), HookError> { let recorded = self.recorded_checkpoints().await?; if sync::lock(recorded).contains(&key) { return Ok(()); @@ -736,18 +802,8 @@ impl FabroHooks { let found = self.workspaces.find(&workspace, key).await; drop(held); let sha = found - .map_err(|error| { - format!( - "the checkpoint commit could not be looked up: {}", - collect_chain(&error).join(": ") - ) - })? - .ok_or_else(|| { - format!( - "no checkpoint commit exists for execution {} firing {} attempt {}", - key.execution, key.firing, key.attempt - ) - })?; + .map_err(HookError::Find)? + .ok_or(HookError::NoCommit { key })?; (workspace, sha) }; let (diff_summary, patch_blob) = match self.stage_diff(&workspace, &sha).await { @@ -758,7 +814,7 @@ impl FabroHooks { run_id = %self.run_id, workspace, sha, - error = %error, + error = %error.render(), "the checkpoint's diff was not computed" ); (None, None) @@ -784,11 +840,9 @@ impl FabroHooks { }), ) .await - .map_err(|error| { - format!( - "the checkpoint record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "checkpoint", + source, })?; sync::lock(recorded).insert(key); self.checkpoints.set_last((workspace, sha)); @@ -803,12 +857,16 @@ impl FabroHooks { &self, workspace: &str, sha: &str, - ) -> Result<(Option, Option), String> { + ) -> Result<(Option, Option), HookError> { + let failed = |source| HookError::Diff { + what: "stage's diff", + source, + }; let parent = self .workspaces .commit_parent(workspace, sha) .await - .map_err(|error| collect_chain(&error).join(": "))?; + .map_err(failed)?; let Some(parent) = parent else { return Ok((None, None)); }; @@ -816,14 +874,14 @@ impl FabroHooks { .workspaces .diff(workspace, Some(&parent), sha) .await - .map_err(|error| collect_chain(&error).join(": "))?; + .map_err(failed)?; let patch_blob = self.patch_blob(&diff).await?; Ok((Some(diff.summary), patch_blob)) } /// The patch of a diff in the blob table, when the diff is not empty /// and the run has a blob table. - async fn patch_blob(&self, diff: &WorkspaceDiff) -> Result, String> { + async fn patch_blob(&self, diff: &WorkspaceDiff) -> Result, HookError> { if diff.is_empty() { return Ok(None); } @@ -834,14 +892,17 @@ impl FabroHooks { .write(diff.patch.as_bytes()) .await .map(Some) - .map_err(|error| format!("the patch could not be stored: {error:#}")) + .map_err(|source| HookError::Blob { + what: "patch".to_string(), + source, + }) } /// The checkpoints already recorded for the run, read once: what a /// resume's reissued routing decisions must not record again, and /// where the run's diff is measured to when this process made no /// checkpoint yet. - async fn recorded_checkpoints(&self) -> Result<&Mutex>, String> { + async fn recorded_checkpoints(&self) -> Result<&Mutex>, HookError> { self.checkpoints .recorded .get_or_try_init(|| async { @@ -849,11 +910,9 @@ impl FabroHooks { .records .read_kind(&self.run_id, PlatformRecordKind::Checkpoint) .await - .map_err(|error| { - format!( - "the run's checkpoint records could not be read: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Read { + kind: "checkpoint", + source, })?; let mut recorded = HashSet::new(); let mut last = None; @@ -889,10 +948,10 @@ impl FabroHooks { context: &HookContext, scope: ScopeId, key: CheckpointKey, - ) -> Result { + ) -> Result { let globs = match &self.artifacts.globs { Ok(globs) => globs, - Err(error) => return Err(format!("invalid run.artifacts.include pattern: {error}")), + Err(error) => return Err(HookError::Globs(Arc::clone(error))), }; if globs.is_empty() { return Ok(0); @@ -903,7 +962,7 @@ impl FabroHooks { return Ok(0); }; let Some(blobs) = &self.blobs else { - return Err("the run has no blob table to collect artifacts into".to_string()); + return Err(HookError::NoBlobs); }; let already = self.collected_artifacts().await?; let candidates = list_artifacts(env.as_ref(), globs).await?; @@ -930,7 +989,10 @@ impl FabroHooks { let blob = blobs .write(&bytes) .await - .map_err(|error| format!("the artifact `{path}` could not be stored: {error:#}"))?; + .map_err(|source| HookError::Blob { + what: format!("artifact `{path}`"), + source, + })?; let record = PlatformRecord::ArtifactCollected(ArtifactCollectedRecord { execution: key.execution, firing: key.firing, @@ -951,11 +1013,9 @@ impl FabroHooks { }), ) .await - .map_err(|error| { - format!( - "the artifact record for `{path}` could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "artifact", + source, })?; sync::lock(already).insert(identity); total_bytes = total_bytes.saturating_add(size); @@ -966,7 +1026,7 @@ impl FabroHooks { /// The artifacts already collected for the run, read once: a file that /// is unchanged since it was collected is not collected again. - async fn collected_artifacts(&self) -> Result<&Mutex>, String> { + async fn collected_artifacts(&self) -> Result<&Mutex>, HookError> { self.artifacts .collected .get_or_try_init(|| async { @@ -974,11 +1034,9 @@ impl FabroHooks { .records .read_kind(&self.run_id, PlatformRecordKind::ArtifactCollected) .await - .map_err(|error| { - format!( - "the run's artifact records could not be read: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Read { + kind: "artifact", + source, })?; let collected = stored .into_iter() @@ -998,7 +1056,7 @@ impl FabroHooks { /// the branch started from, in the snapshot repository on this host. /// Nothing is recorded for a run that never created its branch or /// never checkpointed. - async fn record_run_diff(&self) -> Result<(), String> { + async fn record_run_diff(&self) -> Result<(), HookError> { self.recorded_checkpoints().await?; let branch = match self.checkpoints.branch.get() { Some(branch) => Some(branch.clone()), @@ -1023,11 +1081,9 @@ impl FabroHooks { .workspaces .diff(&workspace, Some(&base_sha), &head_sha) .await - .map_err(|error| { - format!( - "the run's diff could not be computed: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Diff { + what: "run's diff", + source, })?; let patch_blob = self.patch_blob(&diff).await?; let record = PlatformRecord::RunDiff(RunDiffRecord { @@ -1039,11 +1095,9 @@ impl FabroHooks { self.records .append(&self.run_id, &record, None) .await - .map_err(|error| { - format!( - "the run diff record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "run diff", + source, })?; info!( run_id = %self.run_id, @@ -1079,7 +1133,7 @@ impl FabroHooks { async fn list_artifacts( env: &dyn ExecEnv, globs: &WorkspaceGlobSet, -) -> Result, String> { +) -> Result, HookError> { let mut files = Vec::new(); for root in globs.traversal_roots() { let listed = env @@ -1088,8 +1142,9 @@ async fn list_artifacts( ARTIFACT_LIST_DEPTH, ) .await - .map_err(|error| { - format!("the workspace could not be listed below `{root}`: {error}") + .map_err(|source| HookError::List { + root: root.to_string(), + source, })?; for entry in listed { if entry.is_dir { @@ -1176,7 +1231,8 @@ impl ExecutionHooks for FabroHooks { { Ok(Some(note)) => prepared.notes.push(note), Ok(None) => {} - Err(message) => { + Err(error) => { + let message = error.render(); warn!( run_id = %self.run_id, node, @@ -1228,7 +1284,7 @@ impl ExecutionHooks for FabroHooks { error = %problem, "the checkpoint record was not written" ); - problems.push(problem); + problems.push(problem.render()); } match self.collect_artifacts(context, scope, key).await { Ok(0) => {} @@ -1251,7 +1307,7 @@ impl ExecutionHooks for FabroHooks { error = %problem, "artifact collection failed" ); - problems.push(format!("artifact collection failed: {problem}")); + problems.push(format!("artifact collection failed: {}", problem.render())); } } let mut report = self.inner.transition(context, transition).await?; @@ -1267,7 +1323,7 @@ impl ExecutionHooks for FabroHooks { "Petri run finished; recording the run's diff and running the run-end hooks" ); if let Err(error) = self.record_run_diff().await { - warn!(run_id = %self.run_id, error = %error, "the run's diff was not recorded"); + warn!(run_id = %self.run_id, error = %error.render(), "the run's diff was not recorded"); } self.inner.run_finished(context, finished).await } @@ -1305,7 +1361,9 @@ impl ExecutionHooks for FabroHooks { } else { Site::Sandbox(Arc::clone(&acquired.env)) }; - self.restore(&workspace, &site).await + self.restore(&workspace, &site) + .await + .map_err(|error| ScopeAcquiredError::new(error.render())) } } From 84e35076003cf1181d2edd2ee1f07723665d5afb Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 14:59:25 -0400 Subject: [PATCH 05/16] Split the projection fold into a module per source of facts projection.rs held 1,834 lines: the fold state, the platform-record fold, the lifecycle fold, the coordinator fold, the engine and view folds, the progress and Pebble folds, the sandbox mapping, the tool mapping and the model parsing, in one file. VIEWS.md is organised by source; the code now is too. projection/mod.rs keeps RunView, FoldState and the helpers every fold shares; platform.rs, coordinator.rs, engine.rs, progress.rs, sandbox.rs and model.rs each hold one source's rows. A pure move: no function body changed, only the visibility the cross-module calls need. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/projection.rs | 1834 ----------------- .../fabro-petri/src/projection/coordinator.rs | 256 +++ .../fabro-petri/src/projection/engine.rs | 457 ++++ .../fabro-petri/src/projection/mod.rs | 344 ++++ .../fabro-petri/src/projection/model.rs | 41 + .../fabro-petri/src/projection/platform.rs | 344 ++++ .../fabro-petri/src/projection/progress.rs | 423 ++++ .../fabro-petri/src/projection/sandbox.rs | 76 + 8 files changed, 1941 insertions(+), 1834 deletions(-) delete mode 100644 lib/components/fabro-petri/src/projection.rs create mode 100644 lib/components/fabro-petri/src/projection/coordinator.rs create mode 100644 lib/components/fabro-petri/src/projection/engine.rs create mode 100644 lib/components/fabro-petri/src/projection/mod.rs create mode 100644 lib/components/fabro-petri/src/projection/model.rs create mode 100644 lib/components/fabro-petri/src/projection/platform.rs create mode 100644 lib/components/fabro-petri/src/projection/progress.rs create mode 100644 lib/components/fabro-petri/src/projection/sandbox.rs diff --git a/lib/components/fabro-petri/src/projection.rs b/lib/components/fabro-petri/src/projection.rs deleted file mode 100644 index 54183cc9f..000000000 --- a/lib/components/fabro-petri/src/projection.rs +++ /dev/null @@ -1,1834 +0,0 @@ -//! The projection of a Petri run: Petri's public events and Fabro's platform -//! records folded into the view Fabro's read side serves. -//! -//! The fold is pure. [`RunView`] holds the [`RunProjection`] the API serves -//! (`GET /runs/{id}/state`, the run list through its summary) and the -//! bookkeeping the fold needs between items ([`FoldState`]): which Petri -//! firing each stage is, which invocation each execution belongs to and -//! whether it is a parallel branch, which stage asked each open question. -//! Both halves are stored by the projector and reloaded for the next pass, -//! so a pass folds only the items past the committed positions. -//! -//! The mapping follows `VIEWS.md`, row by row. The stage key is `(execution, -//! firing)`; Fabro's `StageId` (`node@visit`) is the display label the -//! `RunProjection` keys stages by, and a label two firings would share (two -//! child invocations with the same node name and visit) is made unique by -//! naming the execution. What the matrix leaves default is left default -//! here and named in the crate's README. -//! -//! Every item the fold sees carries the delivery sequence the projector -//! assigned it (`stream_seq`), which a checkpoint keeps as its `seq`. A -//! stage's `first_event_seq`, the key the stage list sorts by, is not the -//! delivery sequence: two logs' records can be committed in an order that -//! differs from their recording times by a few positions, and the view -//! built live must equal the view rebuilt from the records alone. It is the -//! milliseconds from the run's creation to the stage's `visit.started`, -//! plus one, which is the same however the records were delivered. - -use std::collections::{BTreeMap, BTreeSet}; - -use chrono::{DateTime, TimeZone as _, Utc}; -use fabro_store::platform_records::{ - PlatformRecord, RunLifecycleKind, RunLifecycleRecord, StoredPlatformRecord, -}; -use fabro_types::settings::run::RunEnvironmentSettings; -use fabro_types::{ - BlockedReason, CheckpointRecord as ViewCheckpoint, CodingAgentEvent, CodingEvent, Conclusion, - FailureCategory, FailureDetail, FailureReason, InterviewOption, InterviewQuestionRecord, - ModelRef, ModelUsage, ParallelBranchId, ParallelBranchResult, PendingInterviewRecord, - PullRequestCreation, PullRequestCreationStatus, PullRequestLink, ReviewTarget, - ReviewTargetKind, RunApproval, RunApprovalState, RunArtifact, RunControlAction, RunDiff, - RunFailure, RunId, RunProjection, RunSandbox, RunSandboxFailure, RunSandboxInstance, - RunSandboxPlan, RunSandboxRuntime, RunStatus, RunTiming, SandboxProviderKind, StageCompletion, - StageHandler, StageId, StageInferenceProjection, StageModelUsage, StageOutcome, - StageProjection, StageState, StageTiming, StartRecord, SuccessReason, ToolCategory, ToolSource, - ToolSummary, first_event_seq, format_blob_ref, parse_blob_ref, timing, usage_rollup, -}; -use lithos_llm::catalog::{ModelId, ProviderId}; -use lithos_llm::types::Usage; -use petri_execution::events::{Derived, NodeRef, Parsed, RunEvent, Subject, ViewEvent, WaitState}; -use petri_execution::{CoordinatorEvent, ExecutionId, InvocationId}; -use petri_runtime::engine::{Admission, Event}; -use petri_runtime::ir::{Metrics, SandboxInstance, Status, StepEvent}; -use petri_runtime::steps::QuestionReference; -use serde::{Deserialize, Serialize}; -use serde_json::Value; -use tracing::debug; - -use crate::interview::question_type; - -/// One item the projector hands the fold, with its delivery sequence. -pub enum Item<'a> { - Petri(&'a RunEvent), - Platform(&'a StoredPlatformRecord), -} - -/// A stage as the fold knows it: its label in the projection, and what it -/// learned about it. -#[derive(Clone, Debug, Serialize, Deserialize)] -pub struct StageRef { - pub stage_id: StageId, - /// Whether the stage is a logical one the projection shows, or a - /// lowering node it keeps off the list. - pub shown: bool, - /// The node's instance name and visit, for the collision rule. - pub node_name: String, - pub visit: u32, -} - -/// What the fold knows about one invocation. -#[derive(Clone, Debug, Default, Serialize, Deserialize)] -pub struct InvocationRef { - /// The calling execution and firing, for a nested invocation. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub parent: Option<(u64, u64)>, - /// The parallel group and branch index, for a branch child. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub branch: Option<(StageId, u32)>, - /// The result the invocation recorded, for the root. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub failure: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub output: Option, -} - -/// Whether the run's durable record is whole, as the projector last read it. -#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct RecordHealth { - pub complete: bool, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub incomplete: Vec, -} - -/// The fold's bookkeeping between items. -#[derive(Clone, Debug, Default, Serialize, Deserialize)] -pub struct FoldState { - /// Stages by `":"`. - #[serde(default)] - pub stages: BTreeMap, - /// Labels taken, so a second firing with the same name and visit gets - /// its own. - #[serde(default)] - pub labels: BTreeSet, - #[serde(default)] - pub invocations: BTreeMap, - /// Which invocation each execution belongs to. - #[serde(default)] - pub executions: BTreeMap, - /// Open questions by id: the stage that asked. - #[serde(default)] - pub questions: BTreeMap, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub root: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub started_at: Option, - /// The run's recorded finish, when Petri recorded one. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub finished: Option, - /// The run branch and base sha, when they arrive before `run.started`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub run_branch: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub base_sha: Option, - #[serde(default)] - pub checkpoints: u32, - /// The run's diff as its `run.diff` record gave it, whichever side of - /// the run's finish it arrived on. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub run_diff: Option, - #[serde(default)] - pub health: RecordHealth, - /// Firings (`":"`) whose attempt has recorded a - /// finish: what a position-keyed platform record may be streamed - /// behind. - #[serde(default)] - pub finished_firings: BTreeSet, - /// Whether the run's sandbox still exists after its release - /// (`scope.released` `retained`): kept stopped, or deleted. Absent until - /// the root invocation's lease was released. The view carries the same - /// fact as `RunSandboxInstance.retained`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub sandbox_retained: Option, -} - -impl FoldState { - /// Whether Petri recorded the run's finish. - #[must_use] - pub fn finished_run(&self) -> bool { - self.finished.is_some() - } -} - -/// The view of one run: what the API serves and what the fold keeps. -#[derive(Clone, Debug)] -pub struct RunView { - pub projection: Option, - pub state: FoldState, -} - -impl RunView { - #[must_use] - pub fn new() -> Self { - Self { - projection: None, - state: FoldState::default(), - } - } - - /// Fold one item at its delivery sequence. - pub fn fold(&mut self, item: &Item<'_>, stream_seq: u64) { - match item { - Item::Platform(record) => self.fold_platform(record, stream_seq), - Item::Petri(event) => self.fold_petri(event), - } - } - - /// The run's projection, once its `run.created` record was folded. - #[must_use] - pub fn projection(&self) -> Option<&RunProjection> { - self.projection.as_ref() - } - - // ── Platform records ──────────────────────────────────────────────── - - fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { - let at = millis(stored.recorded_at); - if let PlatformRecord::RunCreated(created) = &stored.record { - let title = created - .title - .clone() - .unwrap_or_else(|| fabro_types::infer_run_title(created.spec.graph.goal())); - let mut projection = RunProjection::new(title, created.spec.clone(), at); - projection.parent_id = created.parent_id; - projection.retried_from = created.retried_from; - projection.web_url.clone_from(&created.web_url); - projection.sandbox = Some(RunSandbox::planned(sandbox_plan( - &projection.spec.settings.run.environment, - ))); - self.projection = Some(projection); - return; - } - let Some(projection) = self.projection.as_mut() else { - debug!( - seq = stored.seq, - kind = %stored.record.kind(), - "platform record before run.created; not folded" - ); - return; - }; - touch(projection, at); - match &stored.record { - PlatformRecord::RunLifecycle(record) => fold_lifecycle(projection, record, at), - PlatformRecord::RunTitle(record) => projection.title.clone_from(&record.title), - PlatformRecord::RunParent(record) => projection.parent_id = record.parent_id, - PlatformRecord::RunArchived => projection.archived_at = Some(at), - PlatformRecord::RunUnarchived => projection.archived_at = None, - PlatformRecord::RunSuperseded(record) => { - projection.superseded_by = Some(record.new_run_id); - } - PlatformRecord::RunCreated(_) - | PlatformRecord::RunNotice(_) - | PlatformRecord::InterviewAnswered(_) - | PlatformRecord::NotificationSent(_) - | PlatformRecord::RunPaired(_) => {} - PlatformRecord::RunBranch(record) => { - self.state.run_branch.clone_from(&record.run_branch); - self.state.base_sha.clone_from(&record.base_sha); - if let Some(start) = projection.start.as_mut() { - start.run_branch.clone_from(&record.run_branch); - start.base_sha.clone_from(&record.base_sha); - } - } - PlatformRecord::GitIdentity(record) => { - projection.git_identity = Some(record.identity.clone()); - } - PlatformRecord::Checkpoint(record) => { - self.state.checkpoints = self.state.checkpoints.saturating_add(1); - let stage = self - .state - .stages - .get(&stage_key(record.execution, record.firing)); - let current_node = stage.map_or_else(String::new, |stage| stage.node_name.clone()); - let stage_id = stage - .filter(|stage| stage.shown) - .map(|stage| stage.stage_id.clone()); - let checkpoint = fabro_types::Checkpoint { - timestamp: at, - current_node: current_node.clone(), - git_commit_sha: record.git_commit_sha.clone(), - }; - // The patch stays in the blob table; the view carries its - // reference for a reader to resolve. - let patch = record.patch_blob.as_ref().map(format_blob_ref); - if let Some(stage) = stage_id.and_then(|stage_id| projection.stage_mut(&stage_id)) { - if patch.is_some() { - stage.diff.clone_from(&patch); - } - } - projection.checkpoints.push(ViewCheckpoint { - seq: u32::try_from(stream_seq).unwrap_or(u32::MAX), - checkpoint, - diff: RunDiff { - patch, - summary: record.diff_summary, - }, - }); - } - PlatformRecord::ArtifactCollected(record) => { - let stage = self - .state - .stages - .get(&stage_key(record.execution, record.firing)); - let Some(stage_id) = stage.map(|stage| stage.stage_id.clone()) else { - debug!( - seq = stored.seq, - path = record.path, - "artifact record for an unknown firing; not folded" - ); - return; - }; - projection.artifacts.push(RunArtifact { - stage_id, - retry: record.attempt, - relative_path: record.path.clone(), - size: record.bytes, - blob: record.blob, - }); - } - PlatformRecord::RunDiff(record) => { - let diff = RunDiff { - patch: record.patch_blob.as_ref().map(format_blob_ref), - summary: record.diff_summary, - }; - if let Some(conclusion) = projection.conclusion.as_mut() { - conclusion.diff = diff.clone(); - } - self.state.run_diff = Some(diff); - } - PlatformRecord::PullRequestRequested(record) => { - projection.pull_request_creation = Some(PullRequestCreation { - id: record.creation_id, - status: PullRequestCreationStatus::Pending, - model: record.model.clone(), - force: record.force, - requested_at: at, - updated_at: at, - pull_request: None, - error: None, - }); - } - PlatformRecord::PullRequestCreated(record) => { - let link = PullRequestLink { - owner: record.owner.clone(), - repo: record.repo.clone(), - number: record.number, - }; - projection.pull_request = Some(link.clone()); - if let Some(creation) = projection - .pull_request_creation - .as_mut() - .filter(|creation| creation.is_pending()) - { - creation.succeed(link, at); - } - } - PlatformRecord::PullRequestFailed(record) => { - if let Some(creation) = - projection - .pull_request_creation - .as_mut() - .filter(|creation| { - creation.is_pending() - && record - .creation_id - .is_none_or(|creation_id| creation_id == creation.id) - }) - { - creation.fail(record.error.clone(), at); - } - } - PlatformRecord::PullRequestLinked(record) => { - let link = record.link(); - projection.pull_request = Some(link.clone()); - if let Some(creation) = projection - .pull_request_creation - .as_mut() - .filter(|creation| creation.is_pending()) - { - creation.succeed(link, at); - } - } - PlatformRecord::PullRequestUnlinked(_) => { - projection.pull_request = None; - projection.pull_request_creation = None; - } - } - } - - // ── Petri events ──────────────────────────────────────────────────── - - fn fold_petri(&mut self, event: &RunEvent) { - let at = millis(event.recorded_at); - if let Some(record) = event.coordinator() { - self.fold_coordinator(record, event, at); - } else if let Some(engine) = event.engine() { - self.fold_engine(engine, event, at); - } else if let Some(view) = event.view() { - self.fold_view(view, event, at); - } - if let Some(projection) = self.projection.as_mut() { - touch(projection, at); - } - } - - fn fold_coordinator(&mut self, record: &CoordinatorEvent, event: &RunEvent, at: DateTime) { - match record { - CoordinatorEvent::RunStarted { - root, forked_from, .. - } => { - self.state.root = Some(root.raw()); - self.state.started_at = Some(event.recorded_at); - if let Some(projection) = self.projection.as_mut() { - // A fork's declaration names its source; a parse failure - // means the source was not a Fabro run, which the - // projection cannot show. - projection.forked_from = forked_from.as_ref().and_then(|origin| { - Some(fabro_types::ForkOrigin { - source_run_id: origin.source.as_str().parse().ok()?, - execution: origin.position.execution.raw(), - firing: origin.position.firing.raw(), - rerun_last: origin.rerun_last, - }) - }); - apply_status(projection, RunStatus::Running, at); - projection.start = Some(StartRecord { - start_time: at, - run_branch: self.state.run_branch.clone(), - base_sha: self.state.base_sha.clone(), - }); - // The scope's sandbox is acquired next; `scope.acquired` - // or `scope.failed` settles it. - if let Some(sandbox) = projection.sandbox.take() { - projection.sandbox = Some(RunSandbox::initializing(sandbox.plan().clone())); - } - } - } - CoordinatorEvent::InvocationDeclared { invocation, .. } => { - let mut info = InvocationRef::default(); - if let Some(parent) = &event.context.parent { - info.parent = Some((parent.execution.raw(), parent.firing.raw())); - if let Some((fork_firing, index)) = branch_slot(&parent.slot) { - let group = self - .state - .stages - .get(&stage_key(parent.execution.raw(), fork_firing)) - .map(|stage| stage.stage_id.clone()); - if let Some(group) = group { - info.branch = Some((group, index)); - } - } - } - self.state.invocations.insert(invocation.raw(), info); - } - CoordinatorEvent::ExecutionDeclared { - execution, - invocation, - .. - } => { - self.state - .executions - .insert(execution.raw(), invocation.raw()); - } - CoordinatorEvent::InvocationFinished { invocation, result } => { - let info = self.state.invocations.entry(invocation.raw()).or_default(); - info.failure = result - .failure - .as_ref() - .map(|failure| failure.message.clone()); - info.output = Some(result.output.clone()); - } - CoordinatorEvent::RunPaused => { - if let Some(projection) = self.projection.as_mut() { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } - } - } - CoordinatorEvent::RunUnpaused => { - if let Some(projection) = self.projection.as_mut() { - let next = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, next, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } - } - } - CoordinatorEvent::RunFinished { status } => { - self.state.finished = Some(status.to_string()); - self.conclude(status.to_string().as_str(), at); - } - // ── Sandbox: the retention outcome (VIEWS.md "Sandbox") ───────── - // The instance stays on `Run.sandbox`: it names what ran, and - // `retained` says whether it still exists. - CoordinatorEvent::ScopeReleased { - invocation, - retained, - .. - } => { - if Some(invocation.raw()) == self.state.root { - self.state.sandbox_retained = Some(*retained); - if let Some(sandbox) = self - .projection - .as_mut() - .and_then(|projection| projection.sandbox.as_mut()) - { - sandbox.set_retained(*retained); - } - } - } - CoordinatorEvent::GraphRegistered { .. } - | CoordinatorEvent::ExecutionFinished { .. } - | CoordinatorEvent::InvocationCancelRequested { .. } - | CoordinatorEvent::RunNoteRecorded { .. } => {} - } - } - - /// The run's conclusion, from its recorded finish and what the stages - /// summed to. - fn conclude(&mut self, status: &str, at: DateTime) { - let Some(projection) = self.projection.as_mut() else { - return; - }; - let root = self - .state - .root - .and_then(|root| self.state.invocations.get(&root)); - let failure_message = root.and_then(|root| root.failure.clone()); - let (run_status, outcome, failure) = match status { - "success" => ( - RunStatus::Succeeded { - reason: SuccessReason::Completed, - }, - StageOutcome::Succeeded, - None, - ), - "cancelled" => ( - RunStatus::Failed { - reason: FailureReason::Cancelled, - }, - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason: FailureReason::Cancelled, - detail: FailureDetail::new( - failure_message - .clone() - .unwrap_or_else(|| "the run was cancelled".to_string()), - FailureCategory::Canceled, - ), - }), - ), - _ => ( - RunStatus::Failed { - reason: FailureReason::WorkflowError, - }, - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason: FailureReason::WorkflowError, - detail: FailureDetail::new( - failure_message - .clone() - .unwrap_or_else(|| "the run failed".to_string()), - FailureCategory::Deterministic, - ), - }), - ), - }; - apply_status(projection, run_status, at); - projection.pending_control = None; - projection.pending_interviews.clear(); - let rollup = usage_rollup::usage_rollup_from_projection(projection); - let (stages, total_retries) = rollup.conclusion_stages(projection); - let wall_time_ms = self.state.started_at.map_or(0, |started| { - u64::try_from(at.timestamp_millis()) - .unwrap_or(0) - .saturating_sub(started) - }); - let timing = RunTiming::new( - wall_time_ms, - rollup.timing.inference_time_ms, - rollup.timing.tool_time_ms, - ); - let last_checkpoint = projection.checkpoints.last(); - projection.conclusion = Some(Conclusion { - timestamp: at, - status: outcome, - timing, - failure, - final_git_commit_sha: last_checkpoint - .and_then(|checkpoint| checkpoint.checkpoint.git_commit_sha.clone()), - stages, - usage: rollup.usage_if_present(), - total_retries, - diff: self - .state - .run_diff - .clone() - .or_else(|| last_checkpoint.map(|checkpoint| checkpoint.diff.clone())) - .unwrap_or_default(), - }); - } - - fn fold_engine(&mut self, engine: &Event, event: &RunEvent, at: DateTime) { - let Some(execution) = event.context.execution else { - return; - }; - match engine { - Event::AdmissionDecided { decision, .. } => { - if let Admission::Skip { outcome } = decision { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.state = StageState::Skipped; - stage.completion = Some(StageCompletion { - outcome: StageOutcome::Skipped, - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - } - } - } - Event::StepStarted { attempt, .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - if attempt.raw() > 1 { - stage.clear_live_timing(); - stage.output = None; - stage.output_bytes = None; - } - stage.state = StageState::Running; - stage.live_streaming = Some(true); - } - } - Event::StepProgressRecorded { ev, .. } => { - self.fold_progress(execution, event, ev, at); - } - Event::StepFinished { - firing, - attempt, - outcome, - } => { - self.state - .finished_firings - .insert(stage_key(execution.raw(), firing.raw())); - let is_final = matches!( - event.derived, - Some(Derived::StepFinished { is_final: true, .. }) - ); - let node_name = event - .subject - .as_ref() - .map(|subject| subject.node.name.to_string()); - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - // The step's output: a string, or a command's `stdout`, - // either of which is a `blob://` reference when the - // step offloaded it. The reference stays as it is; the - // bytes it names are the live log's. - let output = outcome - .output - .as_str() - .or_else(|| outcome.output.get("stdout").and_then(Value::as_str)); - if let Some(output) = output { - if parse_blob_ref(output).is_none() { - stage.output_bytes = Some(output.len() as u64); - } - stage.output = Some(output.to_string()); - } - // A simulated step (a dry run) answers with its text. - let simulated = outcome - .output - .get("simulated") - .and_then(Value::as_bool) - .unwrap_or(false); - if simulated - && matches!( - stage.handler, - Some(StageHandler::Prompt | StageHandler::Agent) - ) - { - if let Some(text) = outcome.output.get("text").and_then(Value::as_str) { - stage.response = Some(text.to_string()); - } - } - // An agent's answer: the `response.` the step wrote - // into the run context, as the prompt step writes it. - if stage.handler == Some(StageHandler::Agent) { - let response = node_name - .as_deref() - .and_then(|name| { - outcome - .context_updates - .get(format!("response.{name}").as_str()) - }) - .and_then(Value::as_str) - .or_else(|| outcome.output.as_str()); - if let Some(response) = response { - stage.response = Some(response.to_string()); - } - } - stage.live_streaming = Some(false); - apply_metrics(stage, &outcome.metrics); - if is_final { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - stage.termination = Some(match outcome.status { - Status::TimedOut => fabro_types::CommandTermination::TimedOut, - Status::Cancelled => fabro_types::CommandTermination::Cancelled, - Status::Success - | Status::PartialSuccess { .. } - | Status::Failure(_) - | Status::Skipped => fabro_types::CommandTermination::Exited, - }); - } else { - stage.state = StageState::Retrying; - debug!(attempt = attempt.raw(), "attempt returned; a retry follows"); - } - } - } - Event::ControlRequested { .. } => { - if let Some(Derived::ControlRequested { - deliverable: true, - answer: Some(answer), - }) = &event.derived - { - let firing_key = event.subject.as_ref().and_then(|subject| { - subject - .firing - .map(|firing| stage_key(execution.raw(), firing.raw())) - }); - self.close_questions(answer.question.as_deref(), firing_key.as_deref(), at); - } - } - // ── Sandbox: the instance (VIEWS.md "Sandbox") ────────────────── - // The run's sandbox is the root invocation's scope. A child - // invocation's scope (a parallel branch) shares or owns another - // one and is not the run's; a re-acquisition (a resume, a - // replaced sandbox) names the current instance. - Event::ScopeAcquired { - sandbox, - duration_ms, - .. - } => { - if let Some(projection) = self.root_scope_projection(event) { - let plan = sandbox_plan_of(projection); - projection.sandbox = Some(RunSandbox::ready( - plan.clone(), - sandbox_instance(&plan, sandbox, *duration_ms), - )); - } - } - Event::ScopeFailed { - provider, - error, - causes, - duration_ms, - .. - } => { - if let Some(projection) = self.root_scope_projection(event) { - let plan = sandbox_plan_of(projection); - let provider = provider - .as_deref() - .and_then(provider_kind) - .unwrap_or_else(|| plan.provider.clone()); - projection.sandbox = Some(RunSandbox::failed(plan, RunSandboxFailure { - provider: provider.to_string(), - error: error.clone(), - causes: causes.clone(), - duration_ms: *duration_ms, - })); - } - } - Event::ExecutionStarted { .. } - | Event::TokenEmitted { .. } - | Event::RoutingResolved { .. } - | Event::RouteApplied { .. } - | Event::RetryElapsed { .. } - | Event::NodeExpanded { .. } - | Event::CancelRequested { .. } - | Event::KillRequested { .. } => {} - } - } - - /// The projection, when `event` is a scope record of the root - /// invocation: the run's own sandbox, not a child invocation's. - fn root_scope_projection(&mut self, event: &RunEvent) -> Option<&mut RunProjection> { - let root = self.state.root?; - if event.context.invocation.map(InvocationId::raw) != Some(root) { - return None; - } - self.projection.as_mut() - } - - fn fold_progress( - &mut self, - execution: ExecutionId, - event: &RunEvent, - ev: &StepEvent, - at: DateTime, - ) { - match ev { - StepEvent::Log { line, .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let output = stage.output.get_or_insert_default(); - output.push_str(line); - output.push('\n'); - stage.output_bytes = Some(output.len() as u64); - stage.live_streaming = Some(true); - } - } - StepEvent::Artifact { .. } => {} - StepEvent::Custom(payload) => { - if let Some(parsed) = event.parsed() { - self.fold_parsed(execution, event, parsed, at); - return; - } - let kind = payload.get("kind").and_then(Value::as_str).unwrap_or(""); - match kind { - "pebble" => self.fold_pebble(execution, event, payload, at), - "attractor.prompt" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.prompt = payload - .get("prompt") - .and_then(Value::as_str) - .map(str::to_string); - let model = payload.get("model").and_then(Value::as_str); - if let Some(model) = model { - let (provider, model_id) = split_model(model); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_PROMPT.to_string(), - provider: provider.map(str::to_string), - model: Some(model_id.to_string()), - reasoning_effort: None, - speed: None, - }); - stage.model = model_ref(provider, model_id); - } - } - } - "attractor.prompt.completed" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.response = payload - .get("response") - .and_then(Value::as_str) - .map(str::to_string); - if let Some(usage) = usage_of(payload.get("usage")) { - stage.usage = usage; - } - } - } - "attractor.fallback.plan" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let route = payload - .get("routes") - .and_then(Value::as_array) - .and_then(|routes| routes.first()); - if let Some(route) = route { - let provider = route.get("provider").and_then(Value::as_str); - let model = route.get("model").and_then(Value::as_str); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.map(str::to_string), - model: model.map(str::to_string), - reasoning_effort: None, - speed: None, - }); - if let Some(model) = model { - stage.model = model_ref(provider, model); - } - } - } - } - // The tools a native session was offered, once per - // session (VIEWS.md "Agent activity", tools available): - // the stage's list is the union over its sessions, by - // name, in the order the sessions listed them. - "attractor.tools" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let tools = payload.get("tools").and_then(Value::as_array); - for tool in tools.into_iter().flatten() { - let Some(summary) = tool_summary(tool) else { - continue; - }; - if !stage - .agent_tools - .iter() - .any(|known| known.name == summary.name) - { - stage.agent_tools.push(summary); - } - } - } - } - "attractor.parallel.branch.started" => { - let invocation = payload.get("invocation").and_then(Value::as_u64); - let index = payload - .get("index") - .and_then(Value::as_u64) - .and_then(|index| u32::try_from(index).ok()); - let fork_firing = payload - .get("occurrence") - .and_then(|occurrence| occurrence.get("firing")) - .and_then(Value::as_u64); - if let (Some(invocation), Some(index), Some(fork_firing)) = - (invocation, index, fork_firing) - { - let group = self - .state - .stages - .get(&stage_key(execution.raw(), fork_firing)) - .map(|stage| stage.stage_id.clone()); - if let Some(group) = group { - self.state.invocations.entry(invocation).or_default().branch = - Some((group, index)); - } - } - } - _ => {} - } - } - } - } - - fn fold_parsed( - &mut self, - execution: ExecutionId, - event: &RunEvent, - parsed: &Parsed, - at: DateTime, - ) { - match parsed { - Parsed::Question { question } => { - let Some(subject) = event.subject.as_ref() else { - return; - }; - let Some(firing) = subject.firing else { - return; - }; - let key = stage_key(execution.raw(), firing.raw()); - let label = self.state.stages.get(&key).map_or_else( - || subject.node.name.to_string(), - |stage| stage.stage_id.to_string(), - ); - self.state.questions.insert(question.id.clone(), key); - let Some(projection) = self.projection.as_mut() else { - return; - }; - projection - .pending_interviews - .insert(question.id.clone(), PendingInterviewRecord { - question: InterviewQuestionRecord { - id: question.id.clone(), - text: question.text.clone(), - stage: label, - question_type: question_type(question), - options: question - .options - .iter() - .map(|option| InterviewOption { - key: option.key.clone(), - label: option.label.clone(), - description: option.description.clone(), - preview: option.preview.clone(), - }) - .collect(), - allow_freeform: question.freeform, - timeout_seconds: question - .timeout_ms - .map(|timeout| timeout as f64 / 1000.0), - context_display: question.context.clone(), - review_target: question.reference.as_ref().and_then(review_target), - }, - started_at: at, - }); - apply_status( - projection, - RunStatus::Blocked { - blocked_reason: BlockedReason::HumanInputRequired, - }, - at, - ); - } - Parsed::QuestionExpired { expired } => { - self.close_questions(Some(expired.question.as_str()), None, at); - } - Parsed::Note { .. } => {} - } - } - - /// Close one question by id, or every question of a firing, and unblock - /// the run when none is left. - fn close_questions( - &mut self, - question: Option<&str>, - firing_key: Option<&str>, - at: DateTime, - ) { - let closed: Vec = match (question, firing_key) { - (Some(question), _) => vec![question.to_string()], - (None, Some(key)) => self - .state - .questions - .iter() - .filter(|(_, asked_by)| asked_by.as_str() == key) - .map(|(id, _)| id.clone()) - .collect(), - (None, None) => Vec::new(), - }; - for id in &closed { - self.state.questions.remove(id); - } - let Some(projection) = self.projection.as_mut() else { - return; - }; - for id in &closed { - projection.pending_interviews.remove(id); - } - if projection.pending_interviews.is_empty() - && matches!(projection.status, RunStatus::Blocked { .. }) - { - apply_status(projection, RunStatus::Running, at); - } - } - - fn fold_pebble( - &mut self, - execution: ExecutionId, - event: &RunEvent, - payload: &Value, - at: DateTime, - ) { - let Some(envelope) = payload.get("event") else { - return; - }; - let envelope: CodingAgentEvent = match serde_json::from_value(envelope.clone()) { - Ok(envelope) => envelope, - Err(error) => { - debug!(error = %error, "a pebble envelope did not decode; skipped"); - return; - } - }; - let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { - return; - }; - let agent = stage.agent.get_or_insert_default(); - agent.apply(&envelope); - if stage.completion.is_none() { - stage.usage = agent.usage.saturating_add(agent.descendant_usage()); - } - // A tool the stage's list names was called, by any of its sessions. - if let CodingEvent::ToolCallStarted { tool_name, .. } = &envelope.event { - if let Some(tool) = stage - .agent_tools - .iter_mut() - .find(|tool| tool.name == *tool_name) - { - tool.invoked = true; - } - } - let is_root = envelope.parent_session_id.is_none(); - #[expect( - clippy::wildcard_enum_match_arm, - reason = "pebble's event vocabulary is non-exhaustive and only some events project" - )] - match &envelope.event { - CodingEvent::SessionStarted { - provider, model, .. - } if is_root => { - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.clone(), - model: model.clone(), - reasoning_effort: None, - speed: None, - }); - if let Some(model) = model.as_deref() { - stage.model = model_ref(provider.as_deref(), model); - } - } - CodingEvent::LlmRequestStarted { requested_model } if is_root => { - stage.inference = Some(StageInferenceProjection { - session_id: envelope.session_id.clone(), - started_at: at, - requested_model: requested_model.clone(), - first_output_at: None, - first_output_kind: None, - retries: 0, - }); - } - CodingEvent::LlmFirstOutput { kind } => { - if let Some(inference) = stage.inference.as_mut() { - if inference.session_id == envelope.session_id { - inference.first_output_at = Some(at); - inference.first_output_kind = Some(*kind); - } - } - } - CodingEvent::LlmRetry { .. } => { - if let Some(inference) = stage.inference.as_mut() { - if inference.session_id == envelope.session_id { - inference.retries = inference.retries.saturating_add(1); - inference.first_output_at = None; - inference.first_output_kind = None; - } - } - } - CodingEvent::AssistantMessage { model, .. } => { - if is_root { - if let Some(provider) = stage - .provider_used - .as_ref() - .and_then(|used| used.provider.as_deref()) - { - stage.model = model_ref(Some(provider), model); - } - } - close_inference(stage, &envelope.session_id, at); - } - CodingEvent::Error { .. } | CodingEvent::RoundInterrupted { .. } => { - close_inference(stage, &envelope.session_id, at); - } - CodingEvent::SessionEnded => { - close_inference(stage, &envelope.session_id, at); - stage.close_tool_batch_for_session(&envelope.session_id, at); - } - CodingEvent::ToolCallStarted { tool_call_id, .. } if is_root => { - stage.open_tool_call(envelope.session_id.clone(), tool_call_id.clone(), at); - } - CodingEvent::ToolCallCompleted { tool_call_id, .. } if is_root => { - stage.close_tool_call(&envelope.session_id, tool_call_id, at); - } - _ => {} - } - } - - fn fold_view(&mut self, view: &ViewEvent, event: &RunEvent, at: DateTime) { - let Some(execution) = event.context.execution else { - return; - }; - match view { - ViewEvent::VisitStarted { .. } => { - let Some(subject) = event.subject.as_ref() else { - return; - }; - self.start_visit(execution, subject, at); - } - ViewEvent::WaitStateChanged { state } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - match state { - WaitState::AwaitingAdmission => { - if stage.state == StageState::Running { - stage.state = StageState::Pending; - } - } - WaitState::Running | WaitState::AwaitingAnswer | WaitState::Cancelling => { - stage.state = StageState::Running; - } - WaitState::AwaitingRetry => stage.state = StageState::Retrying, - } - } - } - ViewEvent::RetryScheduled { .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.state = StageState::Retrying; - } - } - ViewEvent::VisitCompleted { - outcome, - executed, - attempts, - } => { - let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { - return; - }; - stage.state = match outcome.status { - Status::Success => StageState::Succeeded, - Status::PartialSuccess { .. } => StageState::PartiallySucceeded, - Status::Failure(_) | Status::TimedOut => StageState::Failed, - Status::Skipped => StageState::Skipped, - Status::Cancelled => StageState::Cancelled, - }; - if stage.completion.is_none() || !*executed { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - } - if stage.timing.is_none() { - let wall = stage - .started_at - .map_or(0, |started| timing::elapsed_ms(started, at)); - stage.set_authoritative_timing(StageTiming::new(wall, 0, 0)); - } - debug!(attempts, "visit completed"); - } - ViewEvent::ForkCompleted { - occurrence, - results, - .. - } => { - let key = stage_key(occurrence.execution.raw(), occurrence.firing.raw()); - let Some(stage_id) = self - .state - .stages - .get(&key) - .map(|stage| stage.stage_id.clone()) - else { - return; - }; - let Some(projection) = self.projection.as_mut() else { - return; - }; - if let Some(stage) = projection.stage_mut(&stage_id) { - stage.parallel_results = Some( - results - .iter() - .map(|result| ParallelBranchResult { - id: result.node.name.to_string(), - index: Some(result.branch.index as usize), - item_label: None, - status: stage_outcome(&result.status), - context_updates: BTreeMap::new(), - }) - .collect(), - ); - } - } - ViewEvent::ForkStarted { .. } - | ViewEvent::BranchCompleted { .. } - | ViewEvent::RunStalled { .. } => {} - } - } - - /// A firing exists: register its stage and, when it is a logical stage, - /// show it. - fn start_visit(&mut self, execution: ExecutionId, subject: &Subject, at: DateTime) { - let Some(firing) = subject.firing else { - return; - }; - let key = stage_key(execution.raw(), firing.raw()); - if self.state.stages.contains_key(&key) { - return; - } - let node_name = subject.node.name.to_string(); - let visit = visit_of(subject); - let meta_kind = node_meta_kind(&subject.node); - let shown = is_shown(&subject.node); - // Only a shown stage takes a label: a lowering node (a branch's - // parent-side delegate shares its target's name) never competes with - // the stage it stands for. - let mut stage_id = StageId::new(node_name.clone(), visit); - if shown { - stage_id = stage_label(&node_name, visit, execution, &self.state.labels); - self.state.labels.insert(stage_id.to_string()); - } - self.state.stages.insert(key, StageRef { - stage_id: stage_id.clone(), - shown, - node_name, - visit, - }); - if !shown { - return; - } - let branch = self - .state - .executions - .get(&execution.raw()) - .and_then(|invocation| self.state.invocations.get(invocation)) - .and_then(|invocation| invocation.branch.clone()); - let Some(projection) = self.projection.as_mut() else { - return; - }; - let since_created = at - .signed_duration_since(projection.spec.run_id.created_at()) - .num_milliseconds() - .max(0); - let ordinal = u32::try_from(since_created) - .unwrap_or(u32::MAX - 1) - .saturating_add(1); - let stage = projection.stage_entry(stage_id.node_id(), visit, first_event_seq(ordinal)); - stage.handler = Some(StageHandler::from_handler_type(Some(meta_kind))); - stage.started_at = Some(at); - stage.graph_visit = Some(visit); - stage.state = StageState::Pending; - stage.parallel_branch_id = branch.map(|(group, index)| ParallelBranchId::new(group, index)); - } - - /// The shown stage an event's subject firing belongs to. - fn stage_of( - &mut self, - execution: ExecutionId, - subject: Option<&Subject>, - ) -> Option<&mut StageProjection> { - let firing = subject?.firing?; - let stage = self - .state - .stages - .get(&stage_key(execution.raw(), firing.raw()))?; - if !stage.shown { - return None; - } - let stage_id = stage.stage_id.clone(); - self.projection.as_mut()?.stage_mut(&stage_id) - } -} - -impl Default for RunView { - fn default() -> Self { - Self::new() - } -} - -// ── Lifecycle ─────────────────────────────────────────────────────────── - -fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, at: DateTime) { - use RunLifecycleKind as Kind; - match record.transition { - Kind::Submitted => apply_status(projection, RunStatus::Submitted, at), - Kind::StartRequested => {} - Kind::Unpaused => { - let status = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, status, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } - } - Kind::Pending => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - projection.approval = Some(RunApproval { - state: RunApprovalState::Pending, - requested_at: at, - decided_at: None, - denial_reason: None, - }); - } - Kind::Approved => { - if let Some(approval) = projection.approval.as_mut() { - approval.state = RunApprovalState::Approved; - approval.decided_at = Some(at); - } - } - Kind::Denied => { - if let Some(approval) = projection.approval.as_mut() { - approval.state = RunApprovalState::Denied; - approval.decided_at = Some(at); - approval.denial_reason.clone_from(&record.reason); - } - apply_status( - projection, - RunStatus::Failed { - reason: FailureReason::ApprovalDenied, - }, - at, - ); - } - Kind::Runnable => { - // A run left in flight by a restart goes back to the queue: the - // resume's `runnable` steps back from wherever the run stood. - let in_flight = matches!( - projection.status, - RunStatus::Starting - | RunStatus::Running - | RunStatus::Blocked { .. } - | RunStatus::Paused { .. } - ); - if in_flight && record.status == Some(RunStatus::Runnable) { - projection.status = RunStatus::Runnable; - projection.status_updated_at = at; - } else if let Some(status) = record.status { - apply_status(projection, status, at); - } - } - Kind::Blocked => { - // A block that lands while the run is paused waits behind the - // pause: the unpause restores it. - match (projection.status, record.status) { - (RunStatus::Paused { .. }, Some(RunStatus::Blocked { blocked_reason })) => { - apply_status( - projection, - RunStatus::Paused { - prior_block: Some(blocked_reason), - }, - at, - ); - } - (_, Some(status)) => apply_status(projection, status, at), - (_, None) => {} - } - } - Kind::Unblocked => { - let status = match projection.status { - RunStatus::Paused { .. } => RunStatus::Paused { prior_block: None }, - _ => record.status.unwrap_or(RunStatus::Running), - }; - apply_status(projection, status, at); - } - Kind::Starting | Kind::Running | Kind::Removing | Kind::Dead => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - } - Kind::Paused => { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - RunStatus::Paused { prior_block } => prior_block, - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } - } - Kind::Succeeded | Kind::Failed => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - projection.pending_control = None; - if projection.conclusion.is_none() { - let (outcome, failure) = match record.status { - Some(RunStatus::Failed { reason }) => ( - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason, - detail: FailureDetail::new( - record - .reason - .clone() - .unwrap_or_else(|| "the run failed".to_string()), - FailureCategory::Deterministic, - ), - }), - ), - _ => (StageOutcome::Succeeded, None), - }; - projection.conclusion = Some(Conclusion { - timestamp: at, - status: outcome, - timing: RunTiming::default(), - failure, - final_git_commit_sha: None, - stages: Vec::new(), - usage: None, - total_retries: 0, - diff: RunDiff::default(), - }); - } - } - Kind::CancelRequested => projection.pending_control = Some(RunControlAction::Cancel), - Kind::PauseRequested => projection.pending_control = Some(RunControlAction::Pause), - Kind::UnpauseRequested => projection.pending_control = Some(RunControlAction::Unpause), - } -} - -/// Apply a status transition; one the lifecycle refuses is logged and -/// skipped, since the view never fails the run. -fn apply_status(projection: &mut RunProjection, status: RunStatus, at: DateTime) { - if let Err(error) = projection.try_apply_status(status, at) { - debug!(error = %error, "status transition not applied to the Petri projection"); - } -} - -fn touch(projection: &mut RunProjection, at: DateTime) { - if at > projection.last_event_at { - projection.last_event_at = at; - } -} - -// ── Helpers ───────────────────────────────────────────────────────────── - -/// The key of a stage: its execution and firing. -#[must_use] -pub fn stage_key(execution: u64, firing: u64) -> String { - format!("{execution}:{firing}") -} - -/// Which firing of its node a subject is, 1-based. -#[must_use] -pub fn visit_of(subject: &Subject) -> u32 { - subject.visit.unwrap_or(1).max(1) -} - -/// The role a frontend gave a node under `meta.kind`, or the empty string. -fn node_meta_kind(node: &NodeRef) -> &str { - node.meta.get("kind").and_then(Value::as_str).unwrap_or("") -} - -/// Whether a node is a logical stage the projection shows, or a lowering -/// node it keeps off the list: one a frontend marked synthetic, or a -/// parallel branch's delegate. -#[must_use] -pub fn is_shown(node: &NodeRef) -> bool { - let synthetic = node - .meta - .get("synthetic") - .and_then(Value::as_bool) - .unwrap_or(false); - !synthetic && node_meta_kind(node) != "parallel.branch" -} - -/// The label a shown firing takes, which is the stage id the projection -/// keys it by: `node@visit`, or `node/e@visit` when another -/// execution's firing already took that label. `taken` is every label given -/// so far; the caller adds the one returned. The interview adapter labels a -/// question's stage through this same rule, so the stage a question names -/// is the stage the projection shows. -#[must_use] -pub fn stage_label( - node_name: &str, - visit: u32, - execution: ExecutionId, - taken: &BTreeSet, -) -> StageId { - let stage_id = StageId::new(node_name.to_string(), visit); - if taken.contains(&stage_id.to_string()) { - return StageId::new(format!("{node_name}/e{}", execution.raw()), visit); - } - stage_id -} - -/// The fork firing and branch index a branch child's call slot names: -/// `branch:@::`. -fn branch_slot(slot: &str) -> Option<(u64, u32)> { - let rest = slot.strip_prefix("branch:")?; - let mut parts = rest.splitn(3, ':'); - let fork = parts.next()?; - let index = parts.next()?.parse::().ok()?; - let firing = fork.rsplit_once('@')?.1.parse::().ok()?; - Some((firing, index)) -} - -fn millis(recorded_at: u64) -> DateTime { - Utc.timestamp_millis_opt(i64::try_from(recorded_at).unwrap_or(i64::MAX)) - .single() - .unwrap_or_default() -} - -fn sandbox_plan(settings: &RunEnvironmentSettings) -> RunSandboxPlan { - RunSandboxPlan { - provider: settings.provider.clone(), - image: (settings.provider == SandboxProviderKind::DOCKER) - .then(|| settings.image.docker.clone()) - .flatten() - .filter(|image| !image.is_empty()), - snapshot: None, - } -} - -/// The plan the projection's sandbox carries, or the one its environment -/// settings give when no sandbox was projected yet. -fn sandbox_plan_of(projection: &RunProjection) -> RunSandboxPlan { - projection.sandbox.as_ref().map_or_else( - || sandbox_plan(&projection.spec.settings.run.environment), - |sandbox| sandbox.plan().clone(), - ) -} - -/// Fabro's name for the provider Petri's `scope.acquired` names: Petri's -/// `host` is Fabro's `local`; every other kind is spelled the same. `None` -/// for a name that is no provider kind. -fn provider_kind(provider: &str) -> Option { - if provider == "host" { - return Some(SandboxProviderKind::LOCAL); - } - SandboxProviderKind::try_new(provider).ok() -} - -/// The run's sandbox instance from Petri's record of the scope's -/// acquisition: the provider, the provider's id for the sandbox (what a -/// reconnect attaches by), its image and snapshot when the provider knows -/// them, the working directory, and how long the acquisition took. The -/// clone fields stay unset: Petri's checkout copies the bound repository -/// into the workspace and is not a clone Fabro made, and the workspace -/// roots are the provider's own layout, read live. `retained` waits for -/// the scope's release. -fn sandbox_instance( - plan: &RunSandboxPlan, - sandbox: &SandboxInstance, - ready_duration_ms: u64, -) -> RunSandboxInstance { - RunSandboxInstance { - provider: provider_kind(&sandbox.provider) - .unwrap_or_else(|| plan.provider.clone()), - image: sandbox - .image - .as_ref() - .map(ToString::to_string) - .or_else(|| plan.image.clone()), - snapshot: sandbox.snapshot.as_ref().map(ToString::to_string), - runtime: RunSandboxRuntime { - id: sandbox.instance.to_string(), - working_directory: sandbox.working_directory.to_string(), - repo_cloned: None, - clone_origin_url: None, - clone_branch: None, - workspace_root: None, - repos_root: None, - primary_repo_path: None, - primary_repo_link: None, - }, - ready_duration_ms: Some(ready_duration_ms), - retained: None, - } -} - -fn stage_outcome(status: &Status) -> StageOutcome { - match status { - Status::Success => StageOutcome::Succeeded, - Status::PartialSuccess { .. } => StageOutcome::PartiallySucceeded, - Status::Failure(info) => StageOutcome::Failed { - retry_requested: info.class.as_str() == "retry_requested", - }, - Status::Skipped => StageOutcome::Skipped, - Status::Cancelled | Status::TimedOut => StageOutcome::Failed { - retry_requested: false, - }, - } -} - -fn failure_message(status: &Status) -> Option { - match status { - Status::Failure(info) - | Status::PartialSuccess { - underlying: Some(info), - } => Some(info.message.clone()), - Status::TimedOut => Some("the step timed out".to_string()), - Status::Cancelled => Some("the step was cancelled".to_string()), - Status::Success | Status::PartialSuccess { underlying: None } | Status::Skipped => None, - } -} - -/// The finished attempt's metrics onto its stage: the timing and the usage -/// the backend reported. -/// One tool of an `attractor.tools` payload as the stage's list carries -/// it: the name and description as recorded, Pebble's `source` as it is, -/// and Pebble's behavioural category where Petri's says which (a -/// sub-agent tool); every other tool is `other`, because the payload -/// carries Petri's origin category (`builtin`, `mcp`, `host`, `question`), -/// not Pebble's permission class. `invoked` starts false and flips on the -/// session's `ToolCallStarted`. -fn tool_summary(tool: &Value) -> Option { - let name = tool.get("name").and_then(Value::as_str)?; - let description = tool - .get("description") - .and_then(Value::as_str) - .unwrap_or_default(); - let source = tool - .get("source") - .cloned() - .and_then(|source| serde_json::from_value::(source).ok()) - .unwrap_or_default(); - let category = match tool.get("category").and_then(Value::as_str) { - Some("subagent") => ToolCategory::Subagent, - _ => ToolCategory::Other, - }; - Some(ToolSummary { - name: name.to_string(), - description: description.to_string(), - source, - category, - invoked: false, - }) -} - -/// The question's `reference` as Fabro's review target, when it is one -/// Fabro's validation admits (a `document`, or a reference without a kind, -/// with a label and an absolute HTTP URL within Fabro's limits). -fn review_target(reference: &QuestionReference) -> Option { - let kind = match reference.kind.as_deref() { - Some("document") | None => ReviewTargetKind::Document, - Some(_) => return None, - }; - ReviewTarget::new(reference.label.clone(), reference.url.clone(), kind).ok() -} - -fn apply_metrics(stage: &mut StageProjection, metrics: &Metrics) { - let custom = &metrics.custom; - let inference = custom - .get("pebble.inference_ms") - .and_then(Value::as_u64) - .unwrap_or(0); - let tool = custom - .get("pebble.tool_ms") - .and_then(Value::as_u64) - .unwrap_or(0); - let wall = metrics.duration_ms.unwrap_or(0); - let (inference, tool) = match stage.handler { - Some(StageHandler::Prompt) => (wall, 0), - Some(StageHandler::Command) => (0, wall), - _ => (inference, tool), - }; - stage.set_authoritative_timing(StageTiming::new(wall, inference, tool).clamped_to_wall()); - if let Some(usage) = - usage_of(custom.get("pebble.usage")).or_else(|| usage_of(custom.get("prompt.usage"))) - { - stage.usage = usage; - } - if let Some(sessions) = custom - .get("pebble.subagents") - .and_then(|subagents| subagents.get("sessions")) - .and_then(Value::as_array) - { - let mut by_model: Vec = Vec::new(); - for session in sessions { - let provider = session.get("provider").and_then(Value::as_str); - let model = session.get("model").and_then(Value::as_str); - let Some(usage) = usage_of(session.get("usage")) else { - continue; - }; - let Some(model) = model.and_then(|model| model_ref(provider, model)) else { - continue; - }; - if let Some(entry) = by_model.iter_mut().find(|entry| entry.model == model) { - entry.usage = entry.usage.saturating_add(usage); - } else { - by_model.push(ModelUsage::new(model, usage)); - } - } - if !by_model.is_empty() { - stage.usage_by_model = by_model; - } - } -} - -fn usage_of(value: Option<&Value>) -> Option { - serde_json::from_value(value?.clone()).ok() -} - -/// `provider/model` into its parts, or the model alone. -fn split_model(model: &str) -> (Option<&str>, &str) { - match model.split_once('/') { - Some((provider, model)) if !provider.is_empty() && !model.is_empty() => { - (Some(provider), model) - } - _ => (None, model), - } -} - -fn model_ref(provider: Option<&str>, model: &str) -> Option { - let provider = provider.filter(|provider| !provider.is_empty())?; - Some(ModelRef::new( - ProviderId::new(provider), - ModelId::new(model), - )) -} - -fn close_inference(stage: &mut StageProjection, session_id: &str, at: DateTime) { - let open = stage - .inference - .as_ref() - .is_some_and(|inference| inference.session_id == session_id); - if !open { - return; - } - if let Some(inference) = stage.inference.take() { - stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, at)); - } -} - -/// The run id a Petri run key names. -#[must_use] -pub fn run_id_of(key: &str) -> Option { - key.parse().ok() -} - -#[cfg(test)] -mod tests { - use fabro_store::platform_records::RunCreatedRecord; - use fabro_types::test_support as types_support; - use petri_runtime::driver::BranchRole; - use petri_runtime::ir::{FiringId, NodeId}; - - use super::*; - - #[test] - fn a_branch_slot_names_the_fork_firing_and_the_index() { - assert_eq!(branch_slot("branch:fan@7:2:review"), Some((7, 2))); - assert_eq!(branch_slot("branch:fan@7:x:review"), None); - assert_eq!(branch_slot("child:0"), None); - } - - #[test] - fn a_model_selector_splits_into_provider_and_model() { - assert_eq!(split_model("openai/gpt-5.4"), (Some("openai"), "gpt-5.4")); - assert_eq!(split_model("gpt-5.4"), (None, "gpt-5.4")); - assert!(model_ref(None, "gpt-5.4").is_none()); - assert!(model_ref(Some("openai"), "gpt-5.4").is_some()); - } - - #[test] - fn a_taken_label_is_made_unique_by_the_execution() { - let mut view = RunView::new(); - let created = StoredPlatformRecord { - seq: 1, - recorded_at: 1_000, - record: PlatformRecord::RunCreated(RunCreatedRecord { - spec: types_support::test_run_spec(), - title: Some("A run".to_string()), - parent_id: None, - retried_from: None, - web_url: None, - }), - position: None, - }; - view.fold(&Item::Platform(&created), 1); - let subject = |name: &str| Subject { - node: NodeRef { - id: NodeId::new(1), - name: name.into(), - kind: "attractor/command".into(), - meta: serde_json::json!({ "kind": "command" }), - }, - firing: Some(FiringId::new(4)), - visit: Some(1), - attempt: None, - generation: None, - branch: BranchRole::None, - }; - view.start_visit(ExecutionId::new(1), &subject("build"), millis(2_000)); - view.start_visit(ExecutionId::new(2), &subject("build"), millis(3_000)); - let labels: Vec = view - .projection() - .expect("the run was created") - .iter_stages() - .map(|(id, _)| id.to_string()) - .collect(); - assert_eq!(labels, vec!["build@1", "build/e2@1"]); - assert_eq!(view.state.stages.len(), 2); - } -} diff --git a/lib/components/fabro-petri/src/projection/coordinator.rs b/lib/components/fabro-petri/src/projection/coordinator.rs new file mode 100644 index 000000000..ffa5093ce --- /dev/null +++ b/lib/components/fabro-petri/src/projection/coordinator.rs @@ -0,0 +1,256 @@ +//! Petri's coordinator events folded into the view: the run's start, its +//! invocations and executions, pause and unpause, its finish, and the +//! release of its sandbox (VIEWS.md "Run", "Sandbox"). + +use chrono::{DateTime, Utc}; +use fabro_types::{ + Conclusion, FailureCategory, FailureDetail, FailureReason, RunControlAction, RunFailure, + RunSandbox, RunStatus, RunTiming, StageOutcome, StartRecord, SuccessReason, usage_rollup, +}; +use petri_execution::CoordinatorEvent; +use petri_execution::events::RunEvent; + +use super::{InvocationRef, RunView, apply_status, stage_key}; + +impl RunView { + pub(super) fn fold_coordinator( + &mut self, + record: &CoordinatorEvent, + event: &RunEvent, + at: DateTime, + ) { + match record { + CoordinatorEvent::RunStarted { + root, forked_from, .. + } => { + self.state.root = Some(root.raw()); + self.state.started_at = Some(event.recorded_at); + if let Some(projection) = self.projection.as_mut() { + // A fork's declaration names its source; a parse failure + // means the source was not a Fabro run, which the + // projection cannot show. + projection.forked_from = forked_from.as_ref().and_then(|origin| { + Some(fabro_types::ForkOrigin { + source_run_id: origin.source.as_str().parse().ok()?, + execution: origin.position.execution.raw(), + firing: origin.position.firing.raw(), + rerun_last: origin.rerun_last, + }) + }); + apply_status(projection, RunStatus::Running, at); + projection.start = Some(StartRecord { + start_time: at, + run_branch: self.state.run_branch.clone(), + base_sha: self.state.base_sha.clone(), + }); + // The scope's sandbox is acquired next; `scope.acquired` + // or `scope.failed` settles it. + if let Some(sandbox) = projection.sandbox.take() { + projection.sandbox = Some(RunSandbox::initializing(sandbox.plan().clone())); + } + } + } + CoordinatorEvent::InvocationDeclared { invocation, .. } => { + let mut info = InvocationRef::default(); + if let Some(parent) = &event.context.parent { + info.parent = Some((parent.execution.raw(), parent.firing.raw())); + if let Some((fork_firing, index)) = branch_slot(&parent.slot) { + let group = self + .state + .stages + .get(&stage_key(parent.execution.raw(), fork_firing)) + .map(|stage| stage.stage_id.clone()); + if let Some(group) = group { + info.branch = Some((group, index)); + } + } + } + self.state.invocations.insert(invocation.raw(), info); + } + CoordinatorEvent::ExecutionDeclared { + execution, + invocation, + .. + } => { + self.state + .executions + .insert(execution.raw(), invocation.raw()); + } + CoordinatorEvent::InvocationFinished { invocation, result } => { + let info = self.state.invocations.entry(invocation.raw()).or_default(); + info.failure = result + .failure + .as_ref() + .map(|failure| failure.message.clone()); + info.output = Some(result.output.clone()); + } + CoordinatorEvent::RunPaused => { + if let Some(projection) = self.projection.as_mut() { + let prior_block = match projection.status { + RunStatus::Blocked { blocked_reason } => Some(blocked_reason), + _ => None, + }; + apply_status(projection, RunStatus::Paused { prior_block }, at); + if projection.pending_control == Some(RunControlAction::Pause) { + projection.pending_control = None; + } + } + } + CoordinatorEvent::RunUnpaused => { + if let Some(projection) = self.projection.as_mut() { + let next = match projection.status { + RunStatus::Paused { + prior_block: Some(blocked_reason), + } => RunStatus::Blocked { blocked_reason }, + _ => RunStatus::Running, + }; + apply_status(projection, next, at); + if projection.pending_control == Some(RunControlAction::Unpause) { + projection.pending_control = None; + } + } + } + CoordinatorEvent::RunFinished { status } => { + self.state.finished = Some(status.to_string()); + self.conclude(status.to_string().as_str(), at); + } + // ── Sandbox: the retention outcome (VIEWS.md "Sandbox") ───────── + // The instance stays on `Run.sandbox`: it names what ran, and + // `retained` says whether it still exists. + CoordinatorEvent::ScopeReleased { + invocation, + retained, + .. + } => { + if Some(invocation.raw()) == self.state.root { + self.state.sandbox_retained = Some(*retained); + if let Some(sandbox) = self + .projection + .as_mut() + .and_then(|projection| projection.sandbox.as_mut()) + { + sandbox.set_retained(*retained); + } + } + } + CoordinatorEvent::GraphRegistered { .. } + | CoordinatorEvent::ExecutionFinished { .. } + | CoordinatorEvent::InvocationCancelRequested { .. } + | CoordinatorEvent::RunNoteRecorded { .. } => {} + } + } + + /// The run's conclusion, from its recorded finish and what the stages + + /// The run's conclusion, from its recorded finish and what the stages + /// summed to. + fn conclude(&mut self, status: &str, at: DateTime) { + let Some(projection) = self.projection.as_mut() else { + return; + }; + let root = self + .state + .root + .and_then(|root| self.state.invocations.get(&root)); + let failure_message = root.and_then(|root| root.failure.clone()); + let (run_status, outcome, failure) = match status { + "success" => ( + RunStatus::Succeeded { + reason: SuccessReason::Completed, + }, + StageOutcome::Succeeded, + None, + ), + "cancelled" => ( + RunStatus::Failed { + reason: FailureReason::Cancelled, + }, + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason: FailureReason::Cancelled, + detail: FailureDetail::new( + failure_message + .clone() + .unwrap_or_else(|| "the run was cancelled".to_string()), + FailureCategory::Canceled, + ), + }), + ), + _ => ( + RunStatus::Failed { + reason: FailureReason::WorkflowError, + }, + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason: FailureReason::WorkflowError, + detail: FailureDetail::new( + failure_message + .clone() + .unwrap_or_else(|| "the run failed".to_string()), + FailureCategory::Deterministic, + ), + }), + ), + }; + apply_status(projection, run_status, at); + projection.pending_control = None; + projection.pending_interviews.clear(); + let rollup = usage_rollup::usage_rollup_from_projection(projection); + let (stages, total_retries) = rollup.conclusion_stages(projection); + let wall_time_ms = self.state.started_at.map_or(0, |started| { + u64::try_from(at.timestamp_millis()) + .unwrap_or(0) + .saturating_sub(started) + }); + let timing = RunTiming::new( + wall_time_ms, + rollup.timing.inference_time_ms, + rollup.timing.tool_time_ms, + ); + let last_checkpoint = projection.checkpoints.last(); + projection.conclusion = Some(Conclusion { + timestamp: at, + status: outcome, + timing, + failure, + final_git_commit_sha: last_checkpoint + .and_then(|checkpoint| checkpoint.checkpoint.git_commit_sha.clone()), + stages, + usage: rollup.usage_if_present(), + total_retries, + diff: self + .state + .run_diff + .clone() + .or_else(|| last_checkpoint.map(|checkpoint| checkpoint.diff.clone())) + .unwrap_or_default(), + }); + } +} + +/// The fork firing and branch index a branch child's call slot names: +/// `branch:@::`. +fn branch_slot(slot: &str) -> Option<(u64, u32)> { + let rest = slot.strip_prefix("branch:")?; + let mut parts = rest.splitn(3, ':'); + let fork = parts.next()?; + let index = parts.next()?.parse::().ok()?; + let firing = fork.rsplit_once('@')?.1.parse::().ok()?; + Some((firing, index)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_branch_slot_names_the_fork_firing_and_the_index() { + assert_eq!(branch_slot("branch:fan@7:2:review"), Some((7, 2))); + assert_eq!(branch_slot("branch:fan@7:x:review"), None); + assert_eq!(branch_slot("child:0"), None); + } +} diff --git a/lib/components/fabro-petri/src/projection/engine.rs b/lib/components/fabro-petri/src/projection/engine.rs new file mode 100644 index 000000000..a936ab494 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/engine.rs @@ -0,0 +1,457 @@ +//! Petri's engine and view events folded into the stages: a firing's +//! visit, its attempts and their outcomes, its wait states, and the scope +//! its sandbox was acquired in (VIEWS.md "Stages", "Sandbox"). + +use std::collections::BTreeMap; + +use chrono::{DateTime, Utc}; +use fabro_types::{ + ModelUsage, ParallelBranchId, ParallelBranchResult, RunProjection, RunSandbox, + RunSandboxFailure, StageCompletion, StageHandler, StageId, StageOutcome, StageProjection, + StageState, StageTiming, first_event_seq, parse_blob_ref, timing, +}; +use petri_execution::events::{Derived, RunEvent, Subject, ViewEvent, WaitState}; +use petri_execution::{ExecutionId, InvocationId}; +use petri_runtime::engine::{Admission, Event}; +use petri_runtime::ir::{Metrics, Status}; +use serde_json::Value; +use tracing::debug; + +use super::model::{model_ref, usage_of}; +use super::sandbox::{provider_kind, sandbox_instance, sandbox_plan_of}; +use super::{RunView, StageRef, is_shown, node_meta_kind, stage_key, stage_label, visit_of}; + +impl RunView { + pub(super) fn fold_engine(&mut self, engine: &Event, event: &RunEvent, at: DateTime) { + let Some(execution) = event.context.execution else { + return; + }; + match engine { + Event::AdmissionDecided { decision, .. } => { + if let Admission::Skip { outcome } = decision { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.state = StageState::Skipped; + stage.completion = Some(StageCompletion { + outcome: StageOutcome::Skipped, + notes: None, + failure_reason: failure_message(&outcome.status), + timestamp: at, + }); + } + } + } + Event::StepStarted { attempt, .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + if attempt.raw() > 1 { + stage.clear_live_timing(); + stage.output = None; + stage.output_bytes = None; + } + stage.state = StageState::Running; + stage.live_streaming = Some(true); + } + } + Event::StepProgressRecorded { ev, .. } => { + self.fold_progress(execution, event, ev, at); + } + Event::StepFinished { + firing, + attempt, + outcome, + } => { + self.state + .finished_firings + .insert(stage_key(execution.raw(), firing.raw())); + let is_final = matches!( + event.derived, + Some(Derived::StepFinished { is_final: true, .. }) + ); + let node_name = event + .subject + .as_ref() + .map(|subject| subject.node.name.to_string()); + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + // The step's output: a string, or a command's `stdout`, + // either of which is a `blob://` reference when the + // step offloaded it. The reference stays as it is; the + // bytes it names are the live log's. + let output = outcome + .output + .as_str() + .or_else(|| outcome.output.get("stdout").and_then(Value::as_str)); + if let Some(output) = output { + if parse_blob_ref(output).is_none() { + stage.output_bytes = Some(output.len() as u64); + } + stage.output = Some(output.to_string()); + } + // A simulated step (a dry run) answers with its text. + let simulated = outcome + .output + .get("simulated") + .and_then(Value::as_bool) + .unwrap_or(false); + if simulated + && matches!( + stage.handler, + Some(StageHandler::Prompt | StageHandler::Agent) + ) + { + if let Some(text) = outcome.output.get("text").and_then(Value::as_str) { + stage.response = Some(text.to_string()); + } + } + // An agent's answer: the `response.` the step wrote + // into the run context, as the prompt step writes it. + if stage.handler == Some(StageHandler::Agent) { + let response = node_name + .as_deref() + .and_then(|name| { + outcome + .context_updates + .get(format!("response.{name}").as_str()) + }) + .and_then(Value::as_str) + .or_else(|| outcome.output.as_str()); + if let Some(response) = response { + stage.response = Some(response.to_string()); + } + } + stage.live_streaming = Some(false); + apply_metrics(stage, &outcome.metrics); + if is_final { + stage.completion = Some(StageCompletion { + outcome: stage_outcome(&outcome.status), + notes: None, + failure_reason: failure_message(&outcome.status), + timestamp: at, + }); + stage.termination = Some(match outcome.status { + Status::TimedOut => fabro_types::CommandTermination::TimedOut, + Status::Cancelled => fabro_types::CommandTermination::Cancelled, + Status::Success + | Status::PartialSuccess { .. } + | Status::Failure(_) + | Status::Skipped => fabro_types::CommandTermination::Exited, + }); + } else { + stage.state = StageState::Retrying; + debug!(attempt = attempt.raw(), "attempt returned; a retry follows"); + } + } + } + Event::ControlRequested { .. } => { + if let Some(Derived::ControlRequested { + deliverable: true, + answer: Some(answer), + }) = &event.derived + { + let firing_key = event.subject.as_ref().and_then(|subject| { + subject + .firing + .map(|firing| stage_key(execution.raw(), firing.raw())) + }); + self.close_questions(answer.question.as_deref(), firing_key.as_deref(), at); + } + } + // ── Sandbox: the instance (VIEWS.md "Sandbox") ────────────────── + // The run's sandbox is the root invocation's scope. A child + // invocation's scope (a parallel branch) shares or owns another + // one and is not the run's; a re-acquisition (a resume, a + // replaced sandbox) names the current instance. + Event::ScopeAcquired { + sandbox, + duration_ms, + .. + } => { + if let Some(projection) = self.root_scope_projection(event) { + let plan = sandbox_plan_of(projection); + projection.sandbox = Some(RunSandbox::ready( + plan.clone(), + sandbox_instance(&plan, sandbox, *duration_ms), + )); + } + } + Event::ScopeFailed { + provider, + error, + causes, + duration_ms, + .. + } => { + if let Some(projection) = self.root_scope_projection(event) { + let plan = sandbox_plan_of(projection); + let provider = provider + .as_deref() + .and_then(provider_kind) + .unwrap_or_else(|| plan.provider.clone()); + projection.sandbox = Some(RunSandbox::failed(plan, RunSandboxFailure { + provider: provider.to_string(), + error: error.clone(), + causes: causes.clone(), + duration_ms: *duration_ms, + })); + } + } + Event::ExecutionStarted { .. } + | Event::TokenEmitted { .. } + | Event::RoutingResolved { .. } + | Event::RouteApplied { .. } + | Event::RetryElapsed { .. } + | Event::NodeExpanded { .. } + | Event::CancelRequested { .. } + | Event::KillRequested { .. } => {} + } + } + + /// The projection, when `event` is a scope record of the root + + /// The projection, when `event` is a scope record of the root + /// invocation: the run's own sandbox, not a child invocation's. + fn root_scope_projection(&mut self, event: &RunEvent) -> Option<&mut RunProjection> { + let root = self.state.root?; + if event.context.invocation.map(InvocationId::raw) != Some(root) { + return None; + } + self.projection.as_mut() + } + + pub(super) fn fold_view(&mut self, view: &ViewEvent, event: &RunEvent, at: DateTime) { + let Some(execution) = event.context.execution else { + return; + }; + match view { + ViewEvent::VisitStarted { .. } => { + let Some(subject) = event.subject.as_ref() else { + return; + }; + self.start_visit(execution, subject, at); + } + ViewEvent::WaitStateChanged { state } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + match state { + WaitState::AwaitingAdmission => { + if stage.state == StageState::Running { + stage.state = StageState::Pending; + } + } + WaitState::Running | WaitState::AwaitingAnswer | WaitState::Cancelling => { + stage.state = StageState::Running; + } + WaitState::AwaitingRetry => stage.state = StageState::Retrying, + } + } + } + ViewEvent::RetryScheduled { .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.state = StageState::Retrying; + } + } + ViewEvent::VisitCompleted { + outcome, + executed, + attempts, + } => { + let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { + return; + }; + stage.state = match outcome.status { + Status::Success => StageState::Succeeded, + Status::PartialSuccess { .. } => StageState::PartiallySucceeded, + Status::Failure(_) | Status::TimedOut => StageState::Failed, + Status::Skipped => StageState::Skipped, + Status::Cancelled => StageState::Cancelled, + }; + if stage.completion.is_none() || !*executed { + stage.completion = Some(StageCompletion { + outcome: stage_outcome(&outcome.status), + notes: None, + failure_reason: failure_message(&outcome.status), + timestamp: at, + }); + } + if stage.timing.is_none() { + let wall = stage + .started_at + .map_or(0, |started| timing::elapsed_ms(started, at)); + stage.set_authoritative_timing(StageTiming::new(wall, 0, 0)); + } + debug!(attempts, "visit completed"); + } + ViewEvent::ForkCompleted { + occurrence, + results, + .. + } => { + let key = stage_key(occurrence.execution.raw(), occurrence.firing.raw()); + let Some(stage_id) = self + .state + .stages + .get(&key) + .map(|stage| stage.stage_id.clone()) + else { + return; + }; + let Some(projection) = self.projection.as_mut() else { + return; + }; + if let Some(stage) = projection.stage_mut(&stage_id) { + stage.parallel_results = Some( + results + .iter() + .map(|result| ParallelBranchResult { + id: result.node.name.to_string(), + index: Some(result.branch.index as usize), + item_label: None, + status: stage_outcome(&result.status), + context_updates: BTreeMap::new(), + }) + .collect(), + ); + } + } + ViewEvent::ForkStarted { .. } + | ViewEvent::BranchCompleted { .. } + | ViewEvent::RunStalled { .. } => {} + } + } + + /// A firing exists: register its stage and, when it is a logical stage, + + /// A firing exists: register its stage and, when it is a logical stage, + /// show it. + pub(super) fn start_visit( + &mut self, + execution: ExecutionId, + subject: &Subject, + at: DateTime, + ) { + let Some(firing) = subject.firing else { + return; + }; + let key = stage_key(execution.raw(), firing.raw()); + if self.state.stages.contains_key(&key) { + return; + } + let node_name = subject.node.name.to_string(); + let visit = visit_of(subject); + let meta_kind = node_meta_kind(&subject.node); + let shown = is_shown(&subject.node); + // Only a shown stage takes a label: a lowering node (a branch's + // parent-side delegate shares its target's name) never competes with + // the stage it stands for. + let mut stage_id = StageId::new(node_name.clone(), visit); + if shown { + stage_id = stage_label(&node_name, visit, execution, &self.state.labels); + self.state.labels.insert(stage_id.to_string()); + } + self.state.stages.insert(key, StageRef { + stage_id: stage_id.clone(), + shown, + node_name, + visit, + }); + if !shown { + return; + } + let branch = self + .state + .executions + .get(&execution.raw()) + .and_then(|invocation| self.state.invocations.get(invocation)) + .and_then(|invocation| invocation.branch.clone()); + let Some(projection) = self.projection.as_mut() else { + return; + }; + let since_created = at + .signed_duration_since(projection.spec.run_id.created_at()) + .num_milliseconds() + .max(0); + let ordinal = u32::try_from(since_created) + .unwrap_or(u32::MAX - 1) + .saturating_add(1); + let stage = projection.stage_entry(stage_id.node_id(), visit, first_event_seq(ordinal)); + stage.handler = Some(StageHandler::from_handler_type(Some(meta_kind))); + stage.started_at = Some(at); + stage.graph_visit = Some(visit); + stage.state = StageState::Pending; + stage.parallel_branch_id = branch.map(|(group, index)| ParallelBranchId::new(group, index)); + } +} + +pub(super) fn stage_outcome(status: &Status) -> StageOutcome { + match status { + Status::Success => StageOutcome::Succeeded, + Status::PartialSuccess { .. } => StageOutcome::PartiallySucceeded, + Status::Failure(info) => StageOutcome::Failed { + retry_requested: info.class.as_str() == "retry_requested", + }, + Status::Skipped => StageOutcome::Skipped, + Status::Cancelled | Status::TimedOut => StageOutcome::Failed { + retry_requested: false, + }, + } +} + +pub(super) fn failure_message(status: &Status) -> Option { + match status { + Status::Failure(info) + | Status::PartialSuccess { + underlying: Some(info), + } => Some(info.message.clone()), + Status::TimedOut => Some("the step timed out".to_string()), + Status::Cancelled => Some("the step was cancelled".to_string()), + Status::Success | Status::PartialSuccess { underlying: None } | Status::Skipped => None, + } +} + +/// The finished attempt's metrics onto its stage: the timing and the usage +/// the backend reported. +fn apply_metrics(stage: &mut StageProjection, metrics: &Metrics) { + let custom = &metrics.custom; + let inference = custom + .get("pebble.inference_ms") + .and_then(Value::as_u64) + .unwrap_or(0); + let tool = custom + .get("pebble.tool_ms") + .and_then(Value::as_u64) + .unwrap_or(0); + let wall = metrics.duration_ms.unwrap_or(0); + let (inference, tool) = match stage.handler { + Some(StageHandler::Prompt) => (wall, 0), + Some(StageHandler::Command) => (0, wall), + _ => (inference, tool), + }; + stage.set_authoritative_timing(StageTiming::new(wall, inference, tool).clamped_to_wall()); + if let Some(usage) = + usage_of(custom.get("pebble.usage")).or_else(|| usage_of(custom.get("prompt.usage"))) + { + stage.usage = usage; + } + if let Some(sessions) = custom + .get("pebble.subagents") + .and_then(|subagents| subagents.get("sessions")) + .and_then(Value::as_array) + { + let mut by_model: Vec = Vec::new(); + for session in sessions { + let provider = session.get("provider").and_then(Value::as_str); + let model = session.get("model").and_then(Value::as_str); + let Some(usage) = usage_of(session.get("usage")) else { + continue; + }; + let Some(model) = model.and_then(|model| model_ref(provider, model)) else { + continue; + }; + if let Some(entry) = by_model.iter_mut().find(|entry| entry.model == model) { + entry.usage = entry.usage.saturating_add(usage); + } else { + by_model.push(ModelUsage::new(model, usage)); + } + } + if !by_model.is_empty() { + stage.usage_by_model = by_model; + } + } +} diff --git a/lib/components/fabro-petri/src/projection/mod.rs b/lib/components/fabro-petri/src/projection/mod.rs new file mode 100644 index 000000000..4191bdd62 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/mod.rs @@ -0,0 +1,344 @@ +//! The projection of a Petri run: Petri's public events and Fabro's platform +//! records folded into the view Fabro's read side serves. +//! +//! The fold is pure. [`RunView`] holds the [`RunProjection`] the API serves +//! (`GET /runs/{id}/state`, the run list through its summary) and the +//! bookkeeping the fold needs between items ([`FoldState`]): which Petri +//! firing each stage is, which invocation each execution belongs to and +//! whether it is a parallel branch, which stage asked each open question. +//! Both halves are stored by the projector and reloaded for the next pass, +//! so a pass folds only the items past the committed positions. +//! +//! The mapping follows `VIEWS.md`, row by row. The stage key is `(execution, +//! firing)`; Fabro's `StageId` (`node@visit`) is the display label the +//! `RunProjection` keys stages by, and a label two firings would share (two +//! child invocations with the same node name and visit) is made unique by +//! naming the execution. What the matrix leaves default is left default +//! here and named in the crate's README. +//! +//! Every item the fold sees carries the delivery sequence the projector +//! assigned it (`stream_seq`), which a checkpoint keeps as its `seq`. A +//! stage's `first_event_seq`, the key the stage list sorts by, is not the +//! delivery sequence: two logs' records can be committed in an order that +//! differs from their recording times by a few positions, and the view +//! built live must equal the view rebuilt from the records alone. It is the +//! milliseconds from the run's creation to the stage's `visit.started`, +//! plus one, which is the same however the records were delivered. + +mod coordinator; +mod engine; +mod model; +mod platform; +mod progress; +mod sandbox; + +use std::collections::{BTreeMap, BTreeSet}; + +use chrono::{DateTime, TimeZone as _, Utc}; +use fabro_store::platform_records::StoredPlatformRecord; +use fabro_types::{RunDiff, RunId, RunProjection, RunStatus, StageId, StageProjection}; +use petri_execution::ExecutionId; +use petri_execution::events::{NodeRef, RunEvent, Subject}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; +use tracing::debug; + +/// One item the projector hands the fold, with its delivery sequence. +pub enum Item<'a> { + Petri(&'a RunEvent), + Platform(&'a StoredPlatformRecord), +} + +/// A stage as the fold knows it: its label in the projection, and what it +/// learned about it. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct StageRef { + pub stage_id: StageId, + /// Whether the stage is a logical one the projection shows, or a + /// lowering node it keeps off the list. + pub shown: bool, + /// The node's instance name and visit, for the collision rule. + pub node_name: String, + pub visit: u32, +} + +/// What the fold knows about one invocation. +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub struct InvocationRef { + /// The calling execution and firing, for a nested invocation. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub parent: Option<(u64, u64)>, + /// The parallel group and branch index, for a branch child. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub branch: Option<(StageId, u32)>, + /// The result the invocation recorded, for the root. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub failure: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output: Option, +} + +/// Whether the run's durable record is whole, as the projector last read it. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct RecordHealth { + pub complete: bool, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub incomplete: Vec, +} + +/// The fold's bookkeeping between items. +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub struct FoldState { + /// Stages by `":"`. + #[serde(default)] + pub stages: BTreeMap, + /// Labels taken, so a second firing with the same name and visit gets + /// its own. + #[serde(default)] + pub labels: BTreeSet, + #[serde(default)] + pub invocations: BTreeMap, + /// Which invocation each execution belongs to. + #[serde(default)] + pub executions: BTreeMap, + /// Open questions by id: the stage that asked. + #[serde(default)] + pub questions: BTreeMap, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub root: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub started_at: Option, + /// The run's recorded finish, when Petri recorded one. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub finished: Option, + /// The run branch and base sha, when they arrive before `run.started`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub run_branch: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub base_sha: Option, + #[serde(default)] + pub checkpoints: u32, + /// The run's diff as its `run.diff` record gave it, whichever side of + /// the run's finish it arrived on. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub run_diff: Option, + #[serde(default)] + pub health: RecordHealth, + /// Firings (`":"`) whose attempt has recorded a + /// finish: what a position-keyed platform record may be streamed + /// behind. + #[serde(default)] + pub finished_firings: BTreeSet, + /// Whether the run's sandbox still exists after its release + /// (`scope.released` `retained`): kept stopped, or deleted. Absent until + /// the root invocation's lease was released. The view carries the same + /// fact as `RunSandboxInstance.retained`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sandbox_retained: Option, +} + +impl FoldState { + /// Whether Petri recorded the run's finish. + #[must_use] + pub fn finished_run(&self) -> bool { + self.finished.is_some() + } +} + +/// The view of one run: what the API serves and what the fold keeps. +#[derive(Clone, Debug)] +pub struct RunView { + pub projection: Option, + pub state: FoldState, +} + +impl RunView { + #[must_use] + pub fn new() -> Self { + Self { + projection: None, + state: FoldState::default(), + } + } + + /// Fold one item at its delivery sequence. + pub fn fold(&mut self, item: &Item<'_>, stream_seq: u64) { + match item { + Item::Platform(record) => self.fold_platform(record, stream_seq), + Item::Petri(event) => self.fold_petri(event), + } + } + + /// The run's projection, once its `run.created` record was folded. + #[must_use] + pub fn projection(&self) -> Option<&RunProjection> { + self.projection.as_ref() + } + + fn fold_petri(&mut self, event: &RunEvent) { + let at = millis(event.recorded_at); + if let Some(record) = event.coordinator() { + self.fold_coordinator(record, event, at); + } else if let Some(engine) = event.engine() { + self.fold_engine(engine, event, at); + } else if let Some(view) = event.view() { + self.fold_view(view, event, at); + } + if let Some(projection) = self.projection.as_mut() { + touch(projection, at); + } + } + + /// The shown stage an event's subject firing belongs to. + fn stage_of( + &mut self, + execution: ExecutionId, + subject: Option<&Subject>, + ) -> Option<&mut StageProjection> { + let firing = subject?.firing?; + let stage = self + .state + .stages + .get(&stage_key(execution.raw(), firing.raw()))?; + if !stage.shown { + return None; + } + let stage_id = stage.stage_id.clone(); + self.projection.as_mut()?.stage_mut(&stage_id) + } +} + +impl Default for RunView { + fn default() -> Self { + Self::new() + } +} + +// ── Shared by the folds ───────────────────────────────────────────────── + +/// Apply a status transition; one the lifecycle refuses is logged and +/// skipped, since the view never fails the run. +fn apply_status(projection: &mut RunProjection, status: RunStatus, at: DateTime) { + if let Err(error) = projection.try_apply_status(status, at) { + debug!(error = %error, "status transition not applied to the Petri projection"); + } +} + +fn touch(projection: &mut RunProjection, at: DateTime) { + if at > projection.last_event_at { + projection.last_event_at = at; + } +} + +/// The key of a stage: its execution and firing. +#[must_use] +pub fn stage_key(execution: u64, firing: u64) -> String { + format!("{execution}:{firing}") +} + +/// Which firing of its node a subject is, 1-based. +#[must_use] +pub fn visit_of(subject: &Subject) -> u32 { + subject.visit.unwrap_or(1).max(1) +} + +/// The role a frontend gave a node under `meta.kind`, or the empty string. +fn node_meta_kind(node: &NodeRef) -> &str { + node.meta.get("kind").and_then(Value::as_str).unwrap_or("") +} + +/// Whether a node is a logical stage the projection shows, or a lowering +/// node it keeps off the list: one a frontend marked synthetic, or a +/// parallel branch's delegate. +#[must_use] +pub fn is_shown(node: &NodeRef) -> bool { + let synthetic = node + .meta + .get("synthetic") + .and_then(Value::as_bool) + .unwrap_or(false); + !synthetic && node_meta_kind(node) != "parallel.branch" +} + +/// The label a shown firing takes, which is the stage id the projection +/// keys it by: `node@visit`, or `node/e@visit` when another +/// execution's firing already took that label. `taken` is every label given +/// so far; the caller adds the one returned. The interview adapter labels a +/// question's stage through this same rule, so the stage a question names +/// is the stage the projection shows. +#[must_use] +pub fn stage_label( + node_name: &str, + visit: u32, + execution: ExecutionId, + taken: &BTreeSet, +) -> StageId { + let stage_id = StageId::new(node_name.to_string(), visit); + if taken.contains(&stage_id.to_string()) { + return StageId::new(format!("{node_name}/e{}", execution.raw()), visit); + } + stage_id +} + +fn millis(recorded_at: u64) -> DateTime { + Utc.timestamp_millis_opt(i64::try_from(recorded_at).unwrap_or(i64::MAX)) + .single() + .unwrap_or_default() +} + +/// The run id a Petri run key names. +#[must_use] +pub fn run_id_of(key: &str) -> Option { + key.parse().ok() +} + +#[cfg(test)] +mod tests { + use fabro_store::platform_records::{PlatformRecord, RunCreatedRecord}; + use fabro_types::test_support as types_support; + use petri_runtime::driver::BranchRole; + use petri_runtime::ir::{FiringId, NodeId}; + + use super::*; + + #[test] + fn a_taken_label_is_made_unique_by_the_execution() { + let mut view = RunView::new(); + let created = StoredPlatformRecord { + seq: 1, + recorded_at: 1_000, + record: PlatformRecord::RunCreated(RunCreatedRecord { + spec: types_support::test_run_spec(), + title: Some("A run".to_string()), + parent_id: None, + retried_from: None, + web_url: None, + }), + position: None, + }; + view.fold(&Item::Platform(&created), 1); + let subject = |name: &str| Subject { + node: NodeRef { + id: NodeId::new(1), + name: name.into(), + kind: "attractor/command".into(), + meta: serde_json::json!({ "kind": "command" }), + }, + firing: Some(FiringId::new(4)), + visit: Some(1), + attempt: None, + generation: None, + branch: BranchRole::None, + }; + view.start_visit(ExecutionId::new(1), &subject("build"), millis(2_000)); + view.start_visit(ExecutionId::new(2), &subject("build"), millis(3_000)); + let labels: Vec = view + .projection() + .expect("the run was created") + .iter_stages() + .map(|(id, _)| id.to_string()) + .collect(); + assert_eq!(labels, vec!["build@1", "build/e2@1"]); + assert_eq!(view.state.stages.len(), 2); + } +} diff --git a/lib/components/fabro-petri/src/projection/model.rs b/lib/components/fabro-petri/src/projection/model.rs new file mode 100644 index 000000000..20d24a472 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/model.rs @@ -0,0 +1,41 @@ +//! The model and usage facts the records carry, as the view names them. + +use fabro_types::ModelRef; +use lithos_llm::catalog::{ModelId, ProviderId}; +use lithos_llm::types::Usage; +use serde_json::Value; + +pub(super) fn usage_of(value: Option<&Value>) -> Option { + serde_json::from_value(value?.clone()).ok() +} + +/// `provider/model` into its parts, or the model alone. +pub(super) fn split_model(model: &str) -> (Option<&str>, &str) { + match model.split_once('/') { + Some((provider, model)) if !provider.is_empty() && !model.is_empty() => { + (Some(provider), model) + } + _ => (None, model), + } +} + +pub(super) fn model_ref(provider: Option<&str>, model: &str) -> Option { + let provider = provider.filter(|provider| !provider.is_empty())?; + Some(ModelRef::new( + ProviderId::new(provider), + ModelId::new(model), + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_model_selector_splits_into_provider_and_model() { + assert_eq!(split_model("openai/gpt-5.4"), (Some("openai"), "gpt-5.4")); + assert_eq!(split_model("gpt-5.4"), (None, "gpt-5.4")); + assert!(model_ref(None, "gpt-5.4").is_none()); + assert!(model_ref(Some("openai"), "gpt-5.4").is_some()); + } +} diff --git a/lib/components/fabro-petri/src/projection/platform.rs b/lib/components/fabro-petri/src/projection/platform.rs new file mode 100644 index 000000000..d5c592768 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/platform.rs @@ -0,0 +1,344 @@ +//! The platform records folded into the view: the run's creation, its +//! lifecycle, its title and parent, the branch and identity the first +//! checkpoint recorded, every checkpoint and artifact, the run's diff, and +//! the pull request (VIEWS.md "Run", "Checkpoints", "Artifacts", "Pull +//! request"). + +use chrono::{DateTime, Utc}; +use fabro_store::platform_records::{ + PlatformRecord, RunLifecycleKind, RunLifecycleRecord, StoredPlatformRecord, +}; +use fabro_types::{ + CheckpointRecord as ViewCheckpoint, Conclusion, FailureCategory, FailureDetail, FailureReason, + PullRequestCreation, PullRequestCreationStatus, PullRequestLink, RunApproval, RunApprovalState, + RunArtifact, RunControlAction, RunDiff, RunFailure, RunProjection, RunSandbox, RunStatus, + RunTiming, StageOutcome, format_blob_ref, +}; +use tracing::debug; + +use super::sandbox::sandbox_plan; +use super::{RunView, apply_status, millis, stage_key, touch}; + +impl RunView { + pub(super) fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { + let at = millis(stored.recorded_at); + if let PlatformRecord::RunCreated(created) = &stored.record { + let title = created + .title + .clone() + .unwrap_or_else(|| fabro_types::infer_run_title(created.spec.graph.goal())); + let mut projection = RunProjection::new(title, created.spec.clone(), at); + projection.parent_id = created.parent_id; + projection.retried_from = created.retried_from; + projection.web_url.clone_from(&created.web_url); + projection.sandbox = Some(RunSandbox::planned(sandbox_plan( + &projection.spec.settings.run.environment, + ))); + self.projection = Some(projection); + return; + } + let Some(projection) = self.projection.as_mut() else { + debug!( + seq = stored.seq, + kind = %stored.record.kind(), + "platform record before run.created; not folded" + ); + return; + }; + touch(projection, at); + match &stored.record { + PlatformRecord::RunLifecycle(record) => fold_lifecycle(projection, record, at), + PlatformRecord::RunTitle(record) => projection.title.clone_from(&record.title), + PlatformRecord::RunParent(record) => projection.parent_id = record.parent_id, + PlatformRecord::RunArchived => projection.archived_at = Some(at), + PlatformRecord::RunUnarchived => projection.archived_at = None, + PlatformRecord::RunSuperseded(record) => { + projection.superseded_by = Some(record.new_run_id); + } + PlatformRecord::RunCreated(_) + | PlatformRecord::RunNotice(_) + | PlatformRecord::InterviewAnswered(_) + | PlatformRecord::NotificationSent(_) + | PlatformRecord::RunPaired(_) => {} + PlatformRecord::RunBranch(record) => { + self.state.run_branch.clone_from(&record.run_branch); + self.state.base_sha.clone_from(&record.base_sha); + if let Some(start) = projection.start.as_mut() { + start.run_branch.clone_from(&record.run_branch); + start.base_sha.clone_from(&record.base_sha); + } + } + PlatformRecord::GitIdentity(record) => { + projection.git_identity = Some(record.identity.clone()); + } + PlatformRecord::Checkpoint(record) => { + self.state.checkpoints = self.state.checkpoints.saturating_add(1); + let stage = self + .state + .stages + .get(&stage_key(record.execution, record.firing)); + let current_node = stage.map_or_else(String::new, |stage| stage.node_name.clone()); + let stage_id = stage + .filter(|stage| stage.shown) + .map(|stage| stage.stage_id.clone()); + let checkpoint = fabro_types::Checkpoint { + timestamp: at, + current_node: current_node.clone(), + git_commit_sha: record.git_commit_sha.clone(), + }; + // The patch stays in the blob table; the view carries its + // reference for a reader to resolve. + let patch = record.patch_blob.as_ref().map(format_blob_ref); + if let Some(stage) = stage_id.and_then(|stage_id| projection.stage_mut(&stage_id)) { + if patch.is_some() { + stage.diff.clone_from(&patch); + } + } + projection.checkpoints.push(ViewCheckpoint { + seq: u32::try_from(stream_seq).unwrap_or(u32::MAX), + checkpoint, + diff: RunDiff { + patch, + summary: record.diff_summary, + }, + }); + } + PlatformRecord::ArtifactCollected(record) => { + let stage = self + .state + .stages + .get(&stage_key(record.execution, record.firing)); + let Some(stage_id) = stage.map(|stage| stage.stage_id.clone()) else { + debug!( + seq = stored.seq, + path = record.path, + "artifact record for an unknown firing; not folded" + ); + return; + }; + projection.artifacts.push(RunArtifact { + stage_id, + retry: record.attempt, + relative_path: record.path.clone(), + size: record.bytes, + blob: record.blob, + }); + } + PlatformRecord::RunDiff(record) => { + let diff = RunDiff { + patch: record.patch_blob.as_ref().map(format_blob_ref), + summary: record.diff_summary, + }; + if let Some(conclusion) = projection.conclusion.as_mut() { + conclusion.diff = diff.clone(); + } + self.state.run_diff = Some(diff); + } + PlatformRecord::PullRequestRequested(record) => { + projection.pull_request_creation = Some(PullRequestCreation { + id: record.creation_id, + status: PullRequestCreationStatus::Pending, + model: record.model.clone(), + force: record.force, + requested_at: at, + updated_at: at, + pull_request: None, + error: None, + }); + } + PlatformRecord::PullRequestCreated(record) => { + let link = PullRequestLink { + owner: record.owner.clone(), + repo: record.repo.clone(), + number: record.number, + }; + projection.pull_request = Some(link.clone()); + if let Some(creation) = projection + .pull_request_creation + .as_mut() + .filter(|creation| creation.is_pending()) + { + creation.succeed(link, at); + } + } + PlatformRecord::PullRequestFailed(record) => { + if let Some(creation) = + projection + .pull_request_creation + .as_mut() + .filter(|creation| { + creation.is_pending() + && record + .creation_id + .is_none_or(|creation_id| creation_id == creation.id) + }) + { + creation.fail(record.error.clone(), at); + } + } + PlatformRecord::PullRequestLinked(record) => { + let link = record.link(); + projection.pull_request = Some(link.clone()); + if let Some(creation) = projection + .pull_request_creation + .as_mut() + .filter(|creation| creation.is_pending()) + { + creation.succeed(link, at); + } + } + PlatformRecord::PullRequestUnlinked(_) => { + projection.pull_request = None; + projection.pull_request_creation = None; + } + } + } +} + +fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, at: DateTime) { + use RunLifecycleKind as Kind; + match record.transition { + Kind::Submitted => apply_status(projection, RunStatus::Submitted, at), + Kind::StartRequested => {} + Kind::Unpaused => { + let status = match projection.status { + RunStatus::Paused { + prior_block: Some(blocked_reason), + } => RunStatus::Blocked { blocked_reason }, + _ => RunStatus::Running, + }; + apply_status(projection, status, at); + if projection.pending_control == Some(RunControlAction::Unpause) { + projection.pending_control = None; + } + } + Kind::Pending => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + projection.approval = Some(RunApproval { + state: RunApprovalState::Pending, + requested_at: at, + decided_at: None, + denial_reason: None, + }); + } + Kind::Approved => { + if let Some(approval) = projection.approval.as_mut() { + approval.state = RunApprovalState::Approved; + approval.decided_at = Some(at); + } + } + Kind::Denied => { + if let Some(approval) = projection.approval.as_mut() { + approval.state = RunApprovalState::Denied; + approval.decided_at = Some(at); + approval.denial_reason.clone_from(&record.reason); + } + apply_status( + projection, + RunStatus::Failed { + reason: FailureReason::ApprovalDenied, + }, + at, + ); + } + Kind::Runnable => { + // A run left in flight by a restart goes back to the queue: the + // resume's `runnable` steps back from wherever the run stood. + let in_flight = matches!( + projection.status, + RunStatus::Starting + | RunStatus::Running + | RunStatus::Blocked { .. } + | RunStatus::Paused { .. } + ); + if in_flight && record.status == Some(RunStatus::Runnable) { + projection.status = RunStatus::Runnable; + projection.status_updated_at = at; + } else if let Some(status) = record.status { + apply_status(projection, status, at); + } + } + Kind::Blocked => { + // A block that lands while the run is paused waits behind the + // pause: the unpause restores it. + match (projection.status, record.status) { + (RunStatus::Paused { .. }, Some(RunStatus::Blocked { blocked_reason })) => { + apply_status( + projection, + RunStatus::Paused { + prior_block: Some(blocked_reason), + }, + at, + ); + } + (_, Some(status)) => apply_status(projection, status, at), + (_, None) => {} + } + } + Kind::Unblocked => { + let status = match projection.status { + RunStatus::Paused { .. } => RunStatus::Paused { prior_block: None }, + _ => record.status.unwrap_or(RunStatus::Running), + }; + apply_status(projection, status, at); + } + Kind::Starting | Kind::Running | Kind::Removing | Kind::Dead => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + } + Kind::Paused => { + let prior_block = match projection.status { + RunStatus::Blocked { blocked_reason } => Some(blocked_reason), + RunStatus::Paused { prior_block } => prior_block, + _ => None, + }; + apply_status(projection, RunStatus::Paused { prior_block }, at); + if projection.pending_control == Some(RunControlAction::Pause) { + projection.pending_control = None; + } + } + Kind::Succeeded | Kind::Failed => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + projection.pending_control = None; + if projection.conclusion.is_none() { + let (outcome, failure) = match record.status { + Some(RunStatus::Failed { reason }) => ( + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason, + detail: FailureDetail::new( + record + .reason + .clone() + .unwrap_or_else(|| "the run failed".to_string()), + FailureCategory::Deterministic, + ), + }), + ), + _ => (StageOutcome::Succeeded, None), + }; + projection.conclusion = Some(Conclusion { + timestamp: at, + status: outcome, + timing: RunTiming::default(), + failure, + final_git_commit_sha: None, + stages: Vec::new(), + usage: None, + total_retries: 0, + diff: RunDiff::default(), + }); + } + } + Kind::CancelRequested => projection.pending_control = Some(RunControlAction::Cancel), + Kind::PauseRequested => projection.pending_control = Some(RunControlAction::Pause), + Kind::UnpauseRequested => projection.pending_control = Some(RunControlAction::Unpause), + } +} diff --git a/lib/components/fabro-petri/src/projection/progress.rs b/lib/components/fabro-petri/src/projection/progress.rs new file mode 100644 index 000000000..7bfe03428 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/progress.rs @@ -0,0 +1,423 @@ +//! A step's progress records folded into its stage: a command's log lines, +//! the payloads the Attractor steps emit (the prompt and its completion, +//! the fallback plan, the tools a session was offered, a parallel branch's +//! start), Pebble's coding-agent envelope, and a gate's question (VIEWS.md +//! "Agent activity", "Questions"). + +use chrono::{DateTime, Utc}; +use fabro_types::{ + BlockedReason, CodingAgentEvent, CodingEvent, InterviewOption, InterviewQuestionRecord, + PendingInterviewRecord, ReviewTarget, ReviewTargetKind, RunStatus, StageInferenceProjection, + StageModelUsage, StageProjection, ToolCategory, ToolSource, ToolSummary, timing, +}; +use petri_execution::ExecutionId; +use petri_execution::events::{Parsed, RunEvent}; +use petri_runtime::ir::StepEvent; +use petri_runtime::steps::QuestionReference; +use serde_json::Value; +use tracing::debug; + +use super::model::{model_ref, split_model, usage_of}; +use super::{RunView, apply_status, stage_key}; +use crate::interview::question_type; + +impl RunView { + pub(super) fn fold_progress( + &mut self, + execution: ExecutionId, + event: &RunEvent, + ev: &StepEvent, + at: DateTime, + ) { + match ev { + StepEvent::Log { line, .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + let output = stage.output.get_or_insert_default(); + output.push_str(line); + output.push('\n'); + stage.output_bytes = Some(output.len() as u64); + stage.live_streaming = Some(true); + } + } + StepEvent::Artifact { .. } => {} + StepEvent::Custom(payload) => { + if let Some(parsed) = event.parsed() { + self.fold_parsed(execution, event, parsed, at); + return; + } + let kind = payload.get("kind").and_then(Value::as_str).unwrap_or(""); + match kind { + "pebble" => self.fold_pebble(execution, event, payload, at), + "attractor.prompt" => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.prompt = payload + .get("prompt") + .and_then(Value::as_str) + .map(str::to_string); + let model = payload.get("model").and_then(Value::as_str); + if let Some(model) = model { + let (provider, model_id) = split_model(model); + stage.provider_used = Some(StageModelUsage { + mode: StageModelUsage::MODE_PROMPT.to_string(), + provider: provider.map(str::to_string), + model: Some(model_id.to_string()), + reasoning_effort: None, + speed: None, + }); + stage.model = model_ref(provider, model_id); + } + } + } + "attractor.prompt.completed" => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.response = payload + .get("response") + .and_then(Value::as_str) + .map(str::to_string); + if let Some(usage) = usage_of(payload.get("usage")) { + stage.usage = usage; + } + } + } + "attractor.fallback.plan" => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + let route = payload + .get("routes") + .and_then(Value::as_array) + .and_then(|routes| routes.first()); + if let Some(route) = route { + let provider = route.get("provider").and_then(Value::as_str); + let model = route.get("model").and_then(Value::as_str); + stage.provider_used = Some(StageModelUsage { + mode: StageModelUsage::MODE_AGENT.to_string(), + provider: provider.map(str::to_string), + model: model.map(str::to_string), + reasoning_effort: None, + speed: None, + }); + if let Some(model) = model { + stage.model = model_ref(provider, model); + } + } + } + } + // The tools a native session was offered, once per + // session (VIEWS.md "Agent activity", tools available): + // the stage's list is the union over its sessions, by + // name, in the order the sessions listed them. + "attractor.tools" => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + let tools = payload.get("tools").and_then(Value::as_array); + for tool in tools.into_iter().flatten() { + let Some(summary) = tool_summary(tool) else { + continue; + }; + if !stage + .agent_tools + .iter() + .any(|known| known.name == summary.name) + { + stage.agent_tools.push(summary); + } + } + } + } + "attractor.parallel.branch.started" => { + let invocation = payload.get("invocation").and_then(Value::as_u64); + let index = payload + .get("index") + .and_then(Value::as_u64) + .and_then(|index| u32::try_from(index).ok()); + let fork_firing = payload + .get("occurrence") + .and_then(|occurrence| occurrence.get("firing")) + .and_then(Value::as_u64); + if let (Some(invocation), Some(index), Some(fork_firing)) = + (invocation, index, fork_firing) + { + let group = self + .state + .stages + .get(&stage_key(execution.raw(), fork_firing)) + .map(|stage| stage.stage_id.clone()); + if let Some(group) = group { + self.state.invocations.entry(invocation).or_default().branch = + Some((group, index)); + } + } + } + _ => {} + } + } + } + } + + fn fold_parsed( + &mut self, + execution: ExecutionId, + event: &RunEvent, + parsed: &Parsed, + at: DateTime, + ) { + match parsed { + Parsed::Question { question } => { + let Some(subject) = event.subject.as_ref() else { + return; + }; + let Some(firing) = subject.firing else { + return; + }; + let key = stage_key(execution.raw(), firing.raw()); + let label = self.state.stages.get(&key).map_or_else( + || subject.node.name.to_string(), + |stage| stage.stage_id.to_string(), + ); + self.state.questions.insert(question.id.clone(), key); + let Some(projection) = self.projection.as_mut() else { + return; + }; + projection + .pending_interviews + .insert(question.id.clone(), PendingInterviewRecord { + question: InterviewQuestionRecord { + id: question.id.clone(), + text: question.text.clone(), + stage: label, + question_type: question_type(question), + options: question + .options + .iter() + .map(|option| InterviewOption { + key: option.key.clone(), + label: option.label.clone(), + description: option.description.clone(), + preview: option.preview.clone(), + }) + .collect(), + allow_freeform: question.freeform, + timeout_seconds: question + .timeout_ms + .map(|timeout| timeout as f64 / 1000.0), + context_display: question.context.clone(), + review_target: question.reference.as_ref().and_then(review_target), + }, + started_at: at, + }); + apply_status( + projection, + RunStatus::Blocked { + blocked_reason: BlockedReason::HumanInputRequired, + }, + at, + ); + } + Parsed::QuestionExpired { expired } => { + self.close_questions(Some(expired.question.as_str()), None, at); + } + Parsed::Note { .. } => {} + } + } + + /// Close one question by id, or every question of a firing, and unblock + + /// Close one question by id, or every question of a firing, and unblock + /// the run when none is left. + pub(super) fn close_questions( + &mut self, + question: Option<&str>, + firing_key: Option<&str>, + at: DateTime, + ) { + let closed: Vec = match (question, firing_key) { + (Some(question), _) => vec![question.to_string()], + (None, Some(key)) => self + .state + .questions + .iter() + .filter(|(_, asked_by)| asked_by.as_str() == key) + .map(|(id, _)| id.clone()) + .collect(), + (None, None) => Vec::new(), + }; + for id in &closed { + self.state.questions.remove(id); + } + let Some(projection) = self.projection.as_mut() else { + return; + }; + for id in &closed { + projection.pending_interviews.remove(id); + } + if projection.pending_interviews.is_empty() + && matches!(projection.status, RunStatus::Blocked { .. }) + { + apply_status(projection, RunStatus::Running, at); + } + } + + fn fold_pebble( + &mut self, + execution: ExecutionId, + event: &RunEvent, + payload: &Value, + at: DateTime, + ) { + let Some(envelope) = payload.get("event") else { + return; + }; + let envelope: CodingAgentEvent = match serde_json::from_value(envelope.clone()) { + Ok(envelope) => envelope, + Err(error) => { + debug!(error = %error, "a pebble envelope did not decode; skipped"); + return; + } + }; + let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { + return; + }; + let agent = stage.agent.get_or_insert_default(); + agent.apply(&envelope); + if stage.completion.is_none() { + stage.usage = agent.usage.saturating_add(agent.descendant_usage()); + } + // A tool the stage's list names was called, by any of its sessions. + if let CodingEvent::ToolCallStarted { tool_name, .. } = &envelope.event { + if let Some(tool) = stage + .agent_tools + .iter_mut() + .find(|tool| tool.name == *tool_name) + { + tool.invoked = true; + } + } + let is_root = envelope.parent_session_id.is_none(); + #[expect( + clippy::wildcard_enum_match_arm, + reason = "pebble's event vocabulary is non-exhaustive and only some events project" + )] + match &envelope.event { + CodingEvent::SessionStarted { + provider, model, .. + } if is_root => { + stage.provider_used = Some(StageModelUsage { + mode: StageModelUsage::MODE_AGENT.to_string(), + provider: provider.clone(), + model: model.clone(), + reasoning_effort: None, + speed: None, + }); + if let Some(model) = model.as_deref() { + stage.model = model_ref(provider.as_deref(), model); + } + } + CodingEvent::LlmRequestStarted { requested_model } if is_root => { + stage.inference = Some(StageInferenceProjection { + session_id: envelope.session_id.clone(), + started_at: at, + requested_model: requested_model.clone(), + first_output_at: None, + first_output_kind: None, + retries: 0, + }); + } + CodingEvent::LlmFirstOutput { kind } => { + if let Some(inference) = stage.inference.as_mut() { + if inference.session_id == envelope.session_id { + inference.first_output_at = Some(at); + inference.first_output_kind = Some(*kind); + } + } + } + CodingEvent::LlmRetry { .. } => { + if let Some(inference) = stage.inference.as_mut() { + if inference.session_id == envelope.session_id { + inference.retries = inference.retries.saturating_add(1); + inference.first_output_at = None; + inference.first_output_kind = None; + } + } + } + CodingEvent::AssistantMessage { model, .. } => { + if is_root { + if let Some(provider) = stage + .provider_used + .as_ref() + .and_then(|used| used.provider.as_deref()) + { + stage.model = model_ref(Some(provider), model); + } + } + close_inference(stage, &envelope.session_id, at); + } + CodingEvent::Error { .. } | CodingEvent::RoundInterrupted { .. } => { + close_inference(stage, &envelope.session_id, at); + } + CodingEvent::SessionEnded => { + close_inference(stage, &envelope.session_id, at); + stage.close_tool_batch_for_session(&envelope.session_id, at); + } + CodingEvent::ToolCallStarted { tool_call_id, .. } if is_root => { + stage.open_tool_call(envelope.session_id.clone(), tool_call_id.clone(), at); + } + CodingEvent::ToolCallCompleted { tool_call_id, .. } if is_root => { + stage.close_tool_call(&envelope.session_id, tool_call_id, at); + } + _ => {} + } + } +} + +/// One tool of an `attractor.tools` payload as the stage's list carries +/// it: the name and description as recorded, Pebble's `source` as it is, +/// and Pebble's behavioural category where Petri's says which (a +/// sub-agent tool); every other tool is `other`, because the payload +/// carries Petri's origin category (`builtin`, `mcp`, `host`, `question`), +/// not Pebble's permission class. `invoked` starts false and flips on the +/// session's `ToolCallStarted`. +fn tool_summary(tool: &Value) -> Option { + let name = tool.get("name").and_then(Value::as_str)?; + let description = tool + .get("description") + .and_then(Value::as_str) + .unwrap_or_default(); + let source = tool + .get("source") + .cloned() + .and_then(|source| serde_json::from_value::(source).ok()) + .unwrap_or_default(); + let category = match tool.get("category").and_then(Value::as_str) { + Some("subagent") => ToolCategory::Subagent, + _ => ToolCategory::Other, + }; + Some(ToolSummary { + name: name.to_string(), + description: description.to_string(), + source, + category, + invoked: false, + }) +} + +/// The question's `reference` as Fabro's review target, when it is one +/// Fabro's validation admits (a `document`, or a reference without a kind, +/// with a label and an absolute HTTP URL within Fabro's limits). +fn review_target(reference: &QuestionReference) -> Option { + let kind = match reference.kind.as_deref() { + Some("document") | None => ReviewTargetKind::Document, + Some(_) => return None, + }; + ReviewTarget::new(reference.label.clone(), reference.url.clone(), kind).ok() +} + +fn close_inference(stage: &mut StageProjection, session_id: &str, at: DateTime) { + let open = stage + .inference + .as_ref() + .is_some_and(|inference| inference.session_id == session_id); + if !open { + return; + } + if let Some(inference) = stage.inference.take() { + stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, at)); + } +} diff --git a/lib/components/fabro-petri/src/projection/sandbox.rs b/lib/components/fabro-petri/src/projection/sandbox.rs new file mode 100644 index 000000000..61ee6dca9 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/sandbox.rs @@ -0,0 +1,76 @@ +//! The run's sandbox as the view shows it: the plan its settings give, and +//! the instance Petri's scope records name (VIEWS.md "Sandbox"). + +use fabro_types::settings::run::RunEnvironmentSettings; +use fabro_types::{ + RunProjection, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, SandboxProviderKind, +}; +use petri_runtime::ir::SandboxInstance; + +pub(super) fn sandbox_plan(settings: &RunEnvironmentSettings) -> RunSandboxPlan { + RunSandboxPlan { + provider: settings.provider.clone(), + image: (settings.provider == SandboxProviderKind::DOCKER) + .then(|| settings.image.docker.clone()) + .flatten() + .filter(|image| !image.is_empty()), + snapshot: None, + } +} + +/// The plan the projection's sandbox carries, or the one its environment +/// settings give when no sandbox was projected yet. +pub(super) fn sandbox_plan_of(projection: &RunProjection) -> RunSandboxPlan { + projection.sandbox.as_ref().map_or_else( + || sandbox_plan(&projection.spec.settings.run.environment), + |sandbox| sandbox.plan().clone(), + ) +} + +/// Fabro's name for the provider Petri's `scope.acquired` names: Petri's +/// `host` is Fabro's `local`; every other kind is spelled the same. `None` +/// for a name that is no provider kind. +pub(super) fn provider_kind(provider: &str) -> Option { + if provider == "host" { + return Some(SandboxProviderKind::LOCAL); + } + SandboxProviderKind::try_new(provider).ok() +} + +/// The run's sandbox instance from Petri's record of the scope's +/// acquisition: the provider, the provider's id for the sandbox (what a +/// reconnect attaches by), its image and snapshot when the provider knows +/// them, the working directory, and how long the acquisition took. The +/// clone fields stay unset: Petri's checkout copies the bound repository +/// into the workspace and is not a clone Fabro made, and the workspace +/// roots are the provider's own layout, read live. `retained` waits for +/// the scope's release. +pub(super) fn sandbox_instance( + plan: &RunSandboxPlan, + sandbox: &SandboxInstance, + ready_duration_ms: u64, +) -> RunSandboxInstance { + RunSandboxInstance { + provider: provider_kind(&sandbox.provider) + .unwrap_or_else(|| plan.provider.clone()), + image: sandbox + .image + .as_ref() + .map(ToString::to_string) + .or_else(|| plan.image.clone()), + snapshot: sandbox.snapshot.as_ref().map(ToString::to_string), + runtime: RunSandboxRuntime { + id: sandbox.instance.to_string(), + working_directory: sandbox.working_directory.to_string(), + repo_cloned: None, + clone_origin_url: None, + clone_branch: None, + workspace_root: None, + repos_root: None, + primary_repo_path: None, + primary_repo_link: None, + }, + ready_duration_ms: Some(ready_duration_ms), + retained: None, + } +} From 2f8ee2ff3670f4eabf188095c5a8ec2647227060 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:01:14 -0400 Subject: [PATCH 06/16] Name the fold's repeated shapes once: completion, model usage, pause, conclusion Four small duplications in the projection fold: a StageCompletion literal built three times from an attempt's status, a StageModelUsage literal built three times with no request controls, the pause and unpause status arithmetic written four times across the coordinator and lifecycle folds, and a bare Conclusion built beside the full one. Each is now one function: completion() in the engine fold, StageModelUsage::new, RunStatus::paused and RunStatus::unpaused beside blocked_reason (with settle_control for the pending control they clear), and Conclusion::outcome_only. One case reads differently: a coordinator RunPaused that lands on a run already paused behind a block now keeps that block for the unpause, as the lifecycle Paused already did, instead of dropping it. Co-Authored-By: Claude Fable 5.1 --- .../fabro-petri/src/projection/coordinator.rs | 24 +++--------- .../fabro-petri/src/projection/engine.rs | 31 +++++++-------- .../fabro-petri/src/projection/mod.rs | 12 +++++- .../fabro-petri/src/projection/platform.rs | 39 ++++--------------- .../fabro-petri/src/projection/progress.rs | 36 +++++++---------- lib/foundation/fabro-types/src/conclusion.rs | 24 ++++++++++++ .../fabro-types/src/run_projection.rs | 13 +++++++ lib/foundation/fabro-types/src/status.rs | 21 ++++++++++ 8 files changed, 111 insertions(+), 89 deletions(-) diff --git a/lib/components/fabro-petri/src/projection/coordinator.rs b/lib/components/fabro-petri/src/projection/coordinator.rs index ffa5093ce..79d784e4a 100644 --- a/lib/components/fabro-petri/src/projection/coordinator.rs +++ b/lib/components/fabro-petri/src/projection/coordinator.rs @@ -10,7 +10,7 @@ use fabro_types::{ use petri_execution::CoordinatorEvent; use petri_execution::events::RunEvent; -use super::{InvocationRef, RunView, apply_status, stage_key}; +use super::{InvocationRef, RunView, apply_status, settle_control, stage_key}; impl RunView { pub(super) fn fold_coordinator( @@ -86,28 +86,14 @@ impl RunView { } CoordinatorEvent::RunPaused => { if let Some(projection) = self.projection.as_mut() { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } + apply_status(projection, projection.status.paused(), at); + settle_control(projection, RunControlAction::Pause); } } CoordinatorEvent::RunUnpaused => { if let Some(projection) = self.projection.as_mut() { - let next = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, next, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } + apply_status(projection, projection.status.unpaused(), at); + settle_control(projection, RunControlAction::Unpause); } } CoordinatorEvent::RunFinished { status } => { diff --git a/lib/components/fabro-petri/src/projection/engine.rs b/lib/components/fabro-petri/src/projection/engine.rs index a936ab494..e542b816b 100644 --- a/lib/components/fabro-petri/src/projection/engine.rs +++ b/lib/components/fabro-petri/src/projection/engine.rs @@ -32,10 +32,8 @@ impl RunView { if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { stage.state = StageState::Skipped; stage.completion = Some(StageCompletion { - outcome: StageOutcome::Skipped, - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, + outcome: StageOutcome::Skipped, + ..completion(&outcome.status, at) }); } } @@ -120,12 +118,7 @@ impl RunView { stage.live_streaming = Some(false); apply_metrics(stage, &outcome.metrics); if is_final { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); + stage.completion = Some(completion(&outcome.status, at)); stage.termination = Some(match outcome.status { Status::TimedOut => fabro_types::CommandTermination::TimedOut, Status::Cancelled => fabro_types::CommandTermination::Cancelled, @@ -263,12 +256,7 @@ impl RunView { Status::Cancelled => StageState::Cancelled, }; if stage.completion.is_none() || !*executed { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); + stage.completion = Some(completion(&outcome.status, at)); } if stage.timing.is_none() { let wall = stage @@ -405,6 +393,17 @@ pub(super) fn failure_message(status: &Status) -> Option { } } +/// A stage's completion from an attempt's status: its outcome and, for a +/// failure, the message. +fn completion(status: &Status, at: DateTime) -> StageCompletion { + StageCompletion { + outcome: stage_outcome(status), + notes: None, + failure_reason: failure_message(status), + timestamp: at, + } +} + /// The finished attempt's metrics onto its stage: the timing and the usage /// the backend reported. fn apply_metrics(stage: &mut StageProjection, metrics: &Metrics) { diff --git a/lib/components/fabro-petri/src/projection/mod.rs b/lib/components/fabro-petri/src/projection/mod.rs index 4191bdd62..6cc9f2cbf 100644 --- a/lib/components/fabro-petri/src/projection/mod.rs +++ b/lib/components/fabro-petri/src/projection/mod.rs @@ -36,7 +36,9 @@ use std::collections::{BTreeMap, BTreeSet}; use chrono::{DateTime, TimeZone as _, Utc}; use fabro_store::platform_records::StoredPlatformRecord; -use fabro_types::{RunDiff, RunId, RunProjection, RunStatus, StageId, StageProjection}; +use fabro_types::{ + RunControlAction, RunDiff, RunId, RunProjection, RunStatus, StageId, StageProjection, +}; use petri_execution::ExecutionId; use petri_execution::events::{NodeRef, RunEvent, Subject}; use serde::{Deserialize, Serialize}; @@ -230,6 +232,14 @@ fn touch(projection: &mut RunProjection, at: DateTime) { } } +/// A control the run acknowledged: the pending control is cleared when it +/// is the one that landed. +fn settle_control(projection: &mut RunProjection, action: RunControlAction) { + if projection.pending_control == Some(action) { + projection.pending_control = None; + } +} + /// The key of a stage: its execution and firing. #[must_use] pub fn stage_key(execution: u64, firing: u64) -> String { diff --git a/lib/components/fabro-petri/src/projection/platform.rs b/lib/components/fabro-petri/src/projection/platform.rs index d5c592768..ebff0cd24 100644 --- a/lib/components/fabro-petri/src/projection/platform.rs +++ b/lib/components/fabro-petri/src/projection/platform.rs @@ -12,12 +12,12 @@ use fabro_types::{ CheckpointRecord as ViewCheckpoint, Conclusion, FailureCategory, FailureDetail, FailureReason, PullRequestCreation, PullRequestCreationStatus, PullRequestLink, RunApproval, RunApprovalState, RunArtifact, RunControlAction, RunDiff, RunFailure, RunProjection, RunSandbox, RunStatus, - RunTiming, StageOutcome, format_blob_ref, + StageOutcome, format_blob_ref, }; use tracing::debug; use super::sandbox::sandbox_plan; -use super::{RunView, apply_status, millis, stage_key, touch}; +use super::{RunView, apply_status, millis, settle_control, stage_key, touch}; impl RunView { pub(super) fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { @@ -201,16 +201,8 @@ fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, a Kind::Submitted => apply_status(projection, RunStatus::Submitted, at), Kind::StartRequested => {} Kind::Unpaused => { - let status = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, status, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } + apply_status(projection, projection.status.unpaused(), at); + settle_control(projection, RunControlAction::Unpause); } Kind::Pending => { if let Some(status) = record.status { @@ -290,15 +282,8 @@ fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, a } } Kind::Paused => { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - RunStatus::Paused { prior_block } => prior_block, - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } + apply_status(projection, projection.status.paused(), at); + settle_control(projection, RunControlAction::Pause); } Kind::Succeeded | Kind::Failed => { if let Some(status) = record.status { @@ -324,17 +309,7 @@ fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, a ), _ => (StageOutcome::Succeeded, None), }; - projection.conclusion = Some(Conclusion { - timestamp: at, - status: outcome, - timing: RunTiming::default(), - failure, - final_git_commit_sha: None, - stages: Vec::new(), - usage: None, - total_retries: 0, - diff: RunDiff::default(), - }); + projection.conclusion = Some(Conclusion::outcome_only(at, outcome, failure)); } } Kind::CancelRequested => projection.pending_control = Some(RunControlAction::Cancel), diff --git a/lib/components/fabro-petri/src/projection/progress.rs b/lib/components/fabro-petri/src/projection/progress.rs index 7bfe03428..95f8996e5 100644 --- a/lib/components/fabro-petri/src/projection/progress.rs +++ b/lib/components/fabro-petri/src/projection/progress.rs @@ -57,13 +57,11 @@ impl RunView { let model = payload.get("model").and_then(Value::as_str); if let Some(model) = model { let (provider, model_id) = split_model(model); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_PROMPT.to_string(), - provider: provider.map(str::to_string), - model: Some(model_id.to_string()), - reasoning_effort: None, - speed: None, - }); + stage.provider_used = Some(StageModelUsage::new( + StageModelUsage::MODE_PROMPT, + provider.map(str::to_string), + Some(model_id.to_string()), + )); stage.model = model_ref(provider, model_id); } } @@ -88,13 +86,11 @@ impl RunView { if let Some(route) = route { let provider = route.get("provider").and_then(Value::as_str); let model = route.get("model").and_then(Value::as_str); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.map(str::to_string), - model: model.map(str::to_string), - reasoning_effort: None, - speed: None, - }); + stage.provider_used = Some(StageModelUsage::new( + StageModelUsage::MODE_AGENT, + provider.map(str::to_string), + model.map(str::to_string), + )); if let Some(model) = model { stage.model = model_ref(provider, model); } @@ -299,13 +295,11 @@ impl RunView { CodingEvent::SessionStarted { provider, model, .. } if is_root => { - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.clone(), - model: model.clone(), - reasoning_effort: None, - speed: None, - }); + stage.provider_used = Some(StageModelUsage::new( + StageModelUsage::MODE_AGENT, + provider.clone(), + model.clone(), + )); if let Some(model) = model.as_deref() { stage.model = model_ref(provider.as_deref(), model); } diff --git a/lib/foundation/fabro-types/src/conclusion.rs b/lib/foundation/fabro-types/src/conclusion.rs index a06697070..65ec19c23 100644 --- a/lib/foundation/fabro-types/src/conclusion.rs +++ b/lib/foundation/fabro-types/src/conclusion.rs @@ -40,3 +40,27 @@ pub struct Conclusion { #[serde(default)] pub diff: RunDiff, } + +impl Conclusion { + /// A conclusion that records only how the run ended: no timing, stages, + /// usage or diff. What a terminal lifecycle record gives when the + /// engine recorded no finish of its own. + #[must_use] + pub fn outcome_only( + timestamp: DateTime, + status: StageOutcome, + failure: Option, + ) -> Self { + Self { + timestamp, + status, + timing: RunTiming::default(), + failure, + final_git_commit_sha: None, + stages: Vec::new(), + usage: None, + total_retries: 0, + diff: RunDiff::default(), + } + } +} diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs index 339aed897..89a884ff5 100644 --- a/lib/foundation/fabro-types/src/run_projection.rs +++ b/lib/foundation/fabro-types/src/run_projection.rs @@ -122,6 +122,19 @@ impl StageModelUsage { pub const MODE_AGENT: &'static str = "agent"; pub const MODE_ACP: &'static str = "acp"; + /// The usage record of a stage that named its provider and model, with + /// no request controls. + #[must_use] + pub fn new(mode: &str, provider: Option, model: Option) -> Self { + Self { + mode: mode.to_string(), + provider, + model, + reasoning_effort: None, + speed: None, + } + } + /// Build the usage record from a `stage.prompt` event, returning `None` /// when the event carried no model metadata. #[must_use] diff --git a/lib/foundation/fabro-types/src/status.rs b/lib/foundation/fabro-types/src/status.rs index 2b2c63e75..1a7db7061 100644 --- a/lib/foundation/fabro-types/src/status.rs +++ b/lib/foundation/fabro-types/src/status.rs @@ -120,6 +120,27 @@ impl RunStatus { } } + /// The status a pause takes the run to: paused, remembering the block + /// the run was under so the unpause can restore it. + #[must_use] + pub fn paused(self) -> Self { + Self::Paused { + prior_block: self.blocked_reason(), + } + } + + /// The status an unpause takes the run to: back to the block the pause + /// remembered, else running. + #[must_use] + pub fn unpaused(self) -> Self { + match self { + Self::Paused { + prior_block: Some(blocked_reason), + } => Self::Blocked { blocked_reason }, + _ => Self::Running, + } + } + pub fn terminal_status(self) -> Option { match self { Self::Succeeded { reason } => Some(TerminalStatus::Succeeded { reason }), From fd0ae7f219b60329691e69016785eed9a9cb8374 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:03:23 -0400 Subject: [PATCH 07/16] Decode the Attractor progress payloads into typed structs fold_progress matched five payload kinds by string and walked each payload's JSON by hand. Progress is now one internally tagged enum over those kinds, with a struct per payload (PlannedRoute, OfferedTool, ForkOccurrence) as the Attractor steps document them, and a test holds each kind literal to the constant petri_attractor_steps exports, so a rename there fails a test here instead of projecting nothing. The fallback plan's original route carries its reasoning effort and speed in the same lithos-llm types StageModelUsage holds, and the fold now keeps them; it set both to None before. That is visible on the stage's provider_used for an agent stage that ran under a fallback plan. Co-Authored-By: Claude Fable 5.1 --- .../fabro-petri/src/projection/progress.rs | 365 +++++++++++++----- 1 file changed, 261 insertions(+), 104 deletions(-) diff --git a/lib/components/fabro-petri/src/projection/progress.rs b/lib/components/fabro-petri/src/projection/progress.rs index 95f8996e5..09121fac1 100644 --- a/lib/components/fabro-petri/src/projection/progress.rs +++ b/lib/components/fabro-petri/src/projection/progress.rs @@ -10,14 +10,16 @@ use fabro_types::{ PendingInterviewRecord, ReviewTarget, ReviewTargetKind, RunStatus, StageInferenceProjection, StageModelUsage, StageProjection, ToolCategory, ToolSource, ToolSummary, timing, }; +use lithos_llm::types::{ReasoningEffort, Speed, Usage}; use petri_execution::ExecutionId; use petri_execution::events::{Parsed, RunEvent}; use petri_runtime::ir::StepEvent; use petri_runtime::steps::QuestionReference; +use serde::Deserialize; use serde_json::Value; use tracing::debug; -use super::model::{model_ref, split_model, usage_of}; +use super::model::{model_ref, split_model}; use super::{RunView, apply_status, stage_key}; use crate::interview::question_type; @@ -45,17 +47,21 @@ impl RunView { self.fold_parsed(execution, event, parsed, at); return; } - let kind = payload.get("kind").and_then(Value::as_str).unwrap_or(""); - match kind { - "pebble" => self.fold_pebble(execution, event, payload, at), - "attractor.prompt" => { + let progress = match Progress::deserialize(payload) { + Ok(progress) => progress, + Err(error) => { + debug!(error = %error, "a progress payload did not decode; skipped"); + return; + } + }; + match progress { + Progress::Pebble { event: envelope } => { + self.fold_pebble(execution, event, envelope, at); + } + Progress::Prompt { prompt, model } => { if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.prompt = payload - .get("prompt") - .and_then(Value::as_str) - .map(str::to_string); - let model = payload.get("model").and_then(Value::as_str); - if let Some(model) = model { + stage.prompt = prompt; + if let Some(model) = model.as_deref() { let (provider, model_id) = split_model(model); stage.provider_used = Some(StageModelUsage::new( StageModelUsage::MODE_PROMPT, @@ -66,34 +72,21 @@ impl RunView { } } } - "attractor.prompt.completed" => { + Progress::PromptCompleted { response, usage } => { if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.response = payload - .get("response") - .and_then(Value::as_str) - .map(str::to_string); - if let Some(usage) = usage_of(payload.get("usage")) { + stage.response = response; + if let Some(usage) = usage { stage.usage = usage; } } } - "attractor.fallback.plan" => { + // `routes[0]` is the original route: what the stage was + // asked to run on, with its request controls. + Progress::FallbackPlan { routes } => { if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let route = payload - .get("routes") - .and_then(Value::as_array) - .and_then(|routes| routes.first()); - if let Some(route) = route { - let provider = route.get("provider").and_then(Value::as_str); - let model = route.get("model").and_then(Value::as_str); - stage.provider_used = Some(StageModelUsage::new( - StageModelUsage::MODE_AGENT, - provider.map(str::to_string), - model.map(str::to_string), - )); - if let Some(model) = model { - stage.model = model_ref(provider, model); - } + if let Some(route) = routes.into_iter().next() { + stage.model = model_ref(Some(&route.provider), &route.model); + stage.provider_used = Some(route.usage()); } } } @@ -101,48 +94,35 @@ impl RunView { // session (VIEWS.md "Agent activity", tools available): // the stage's list is the union over its sessions, by // name, in the order the sessions listed them. - "attractor.tools" => { + Progress::Tools { tools } => { if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let tools = payload.get("tools").and_then(Value::as_array); - for tool in tools.into_iter().flatten() { - let Some(summary) = tool_summary(tool) else { - continue; - }; + for tool in tools { if !stage .agent_tools .iter() - .any(|known| known.name == summary.name) + .any(|known| known.name == tool.name) { - stage.agent_tools.push(summary); + stage.agent_tools.push(tool.summary()); } } } } - "attractor.parallel.branch.started" => { - let invocation = payload.get("invocation").and_then(Value::as_u64); - let index = payload - .get("index") - .and_then(Value::as_u64) - .and_then(|index| u32::try_from(index).ok()); - let fork_firing = payload - .get("occurrence") - .and_then(|occurrence| occurrence.get("firing")) - .and_then(Value::as_u64); - if let (Some(invocation), Some(index), Some(fork_firing)) = - (invocation, index, fork_firing) - { - let group = self - .state - .stages - .get(&stage_key(execution.raw(), fork_firing)) - .map(|stage| stage.stage_id.clone()); - if let Some(group) = group { - self.state.invocations.entry(invocation).or_default().branch = - Some((group, index)); - } + Progress::BranchStarted { + invocation, + index, + occurrence, + } => { + let group = self + .state + .stages + .get(&stage_key(execution.raw(), occurrence.firing)) + .map(|stage| stage.stage_id.clone()); + if let Some(group) = group { + self.state.invocations.entry(invocation).or_default().branch = + Some((group, index)); } } - _ => {} + Progress::Other => {} } } } @@ -255,19 +235,9 @@ impl RunView { &mut self, execution: ExecutionId, event: &RunEvent, - payload: &Value, + envelope: CodingAgentEvent, at: DateTime, ) { - let Some(envelope) = payload.get("event") else { - return; - }; - let envelope: CodingAgentEvent = match serde_json::from_value(envelope.clone()) { - Ok(envelope) => envelope, - Err(error) => { - debug!(error = %error, "a pebble envelope did not decode; skipped"); - return; - } - }; let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { return; }; @@ -361,35 +331,129 @@ impl RunView { } } -/// One tool of an `attractor.tools` payload as the stage's list carries -/// it: the name and description as recorded, Pebble's `source` as it is, -/// and Pebble's behavioural category where Petri's says which (a -/// sub-agent tool); every other tool is `other`, because the payload -/// carries Petri's origin category (`builtin`, `mcp`, `host`, `question`), -/// not Pebble's permission class. `invoked` starts false and flips on the -/// session's `ToolCallStarted`. -fn tool_summary(tool: &Value) -> Option { - let name = tool.get("name").and_then(Value::as_str)?; - let description = tool - .get("description") - .and_then(Value::as_str) - .unwrap_or_default(); - let source = tool - .get("source") - .cloned() - .and_then(|source| serde_json::from_value::(source).ok()) - .unwrap_or_default(); - let category = match tool.get("category").and_then(Value::as_str) { - Some("subagent") => ToolCategory::Subagent, - _ => ToolCategory::Other, - }; - Some(ToolSummary { - name: name.to_string(), - description: description.to_string(), - source, - category, - invoked: false, - }) +/// The `StepEvent::Custom` payloads the fold reads, by their `kind`. The +/// kinds are Petri's, and a test holds each literal to the constant the +/// Attractor steps export, so a rename there fails here rather than +/// projecting nothing. A payload of another kind, or of no kind, is +/// `Other`. +#[derive(Deserialize)] +#[serde(tag = "kind")] +enum Progress { + /// Pebble's coding-agent envelope, forwarded by the agent step. + #[serde(rename = "pebble")] + Pebble { event: CodingAgentEvent }, + /// The prompt step before its first model call: the prompt, and the + /// `provider/model` selector it runs on. + #[serde(rename = "attractor.prompt")] + Prompt { + #[serde(default)] + prompt: Option, + #[serde(default)] + model: Option, + }, + /// The prompt step after its last model call. + #[serde(rename = "attractor.prompt.completed")] + PromptCompleted { + #[serde(default)] + response: Option, + #[serde(default)] + usage: Option, + }, + /// A stage's fallback plan, once per stage: `routes[0]` is the + /// original route. + #[serde(rename = "attractor.fallback.plan")] + FallbackPlan { + #[serde(default)] + routes: Vec, + }, + /// The tools a native session was offered, once per session. + #[serde(rename = "attractor.tools")] + Tools { + #[serde(default)] + tools: Vec, + }, + /// A parallel branch's child started: which fork visit it belongs to, + /// its index, and the child invocation. + #[serde(rename = "attractor.parallel.branch.started")] + BranchStarted { + invocation: u64, + index: u32, + occurrence: ForkOccurrence, + }, + #[serde(other)] + Other, +} + +/// One route of a fallback plan: the provider and model, with the request +/// controls the route carries. +#[derive(Deserialize)] +struct PlannedRoute { + provider: String, + model: String, + #[serde(default)] + reasoning_effort: Option, + #[serde(default)] + speed: Option, +} + +impl PlannedRoute { + /// The route as the stage's model usage: an agent route with its + /// controls. + fn usage(self) -> StageModelUsage { + StageModelUsage { + reasoning_effort: self.reasoning_effort, + speed: self.speed, + ..StageModelUsage::new( + StageModelUsage::MODE_AGENT, + Some(self.provider), + Some(self.model), + ) + } + } +} + +/// The fork visit a branch belongs to: the fork step's firing in the +/// branch's execution. +#[derive(Deserialize)] +struct ForkOccurrence { + firing: u64, +} + +/// One tool of an `attractor.tools` payload: the name and description as +/// recorded, Pebble's `source` as it is, and Petri's origin category +/// (`builtin`, `mcp`, `host`, `question`, `subagent`). +#[derive(Deserialize)] +struct OfferedTool { + name: String, + #[serde(default)] + description: String, + /// Left as recorded: a source Pebble adds later still lists the tool, + /// under the default source. + #[serde(default)] + source: Value, + #[serde(default)] + category: Option, +} + +impl OfferedTool { + /// The tool as the stage's list carries it. Pebble's behavioural + /// category is kept where Petri's says which (a sub-agent tool); every + /// other tool is `other`, because the payload carries Petri's origin + /// category, not Pebble's permission class. `invoked` starts false and + /// flips on the session's `ToolCallStarted`. + fn summary(self) -> ToolSummary { + let category = match self.category.as_deref() { + Some("subagent") => ToolCategory::Subagent, + _ => ToolCategory::Other, + }; + ToolSummary { + name: self.name, + description: self.description, + source: serde_json::from_value::(self.source).unwrap_or_default(), + category, + invoked: false, + } + } } /// The question's `reference` as Fabro's review target, when it is one @@ -415,3 +479,96 @@ fn close_inference(stage: &mut StageProjection, session_id: &str, at: DateTime &'static str { + match Progress::deserialize(&value).expect("a known kind decodes") { + Progress::Pebble { .. } => "pebble", + Progress::Prompt { .. } => "prompt", + Progress::PromptCompleted { .. } => "prompt.completed", + Progress::FallbackPlan { .. } => "fallback.plan", + Progress::Tools { .. } => "tools", + Progress::BranchStarted { .. } => "branch.started", + Progress::Other => "other", + } + }; + assert_eq!(kind_of(json!({ "kind": prompt::PROMPT_EVENT })), "prompt"); + assert_eq!( + kind_of(json!({ "kind": prompt::COMPLETED_EVENT })), + "prompt.completed" + ); + assert_eq!( + kind_of(json!({ "kind": fallback::PLAN_EVENT })), + "fallback.plan" + ); + assert_eq!(kind_of(json!({ "kind": pebble::tools::EVENT })), "tools"); + assert_eq!( + kind_of(json!({ + "kind": parallel::BRANCH_STARTED_EVENT, + "invocation": 3, + "index": 1, + "occurrence": { "fork": "fan", "firing": 7 }, + })), + "branch.started" + ); + assert_eq!( + kind_of(json!({ "kind": parallel::BRANCH_COMPLETED_EVENT })), + "other" + ); + } + + /// A plan's original route carries its request controls onto the + /// stage's model usage. + #[test] + fn a_fallback_plan_route_keeps_its_controls() { + let progress = Progress::deserialize(&json!({ + "kind": fallback::PLAN_EVENT, + "routes": [ + { "position": 0, "provider": "openai", "model": "gpt-5.4", + "reasoning_effort": "high", "speed": null }, + { "position": 1, "provider": "anthropic", "model": "claude" }, + ], + })) + .expect("the plan decodes"); + let Progress::FallbackPlan { routes } = progress else { + panic!("not a plan"); + }; + let usage = routes.into_iter().next().expect("a route").usage(); + assert_eq!(usage.mode, StageModelUsage::MODE_AGENT); + assert_eq!(usage.provider.as_deref(), Some("openai")); + assert_eq!(usage.model.as_deref(), Some("gpt-5.4")); + assert_eq!(usage.reasoning_effort, Some(ReasoningEffort::High)); + assert_eq!(usage.speed, None); + } + + /// A tool lists under Petri's category, with its source as recorded. + #[test] + fn an_offered_tool_maps_to_its_summary() { + let progress = Progress::deserialize(&json!({ + "kind": pebble::tools::EVENT, + "tools": [ + { "name": "spawn_agent", "description": "a child", "source": "native", + "category": "subagent" }, + { "name": "Read", "category": "builtin" }, + ], + })) + .expect("the tools decode"); + let Progress::Tools { tools } = progress else { + panic!("not a tool list"); + }; + let summaries: Vec = tools.into_iter().map(OfferedTool::summary).collect(); + assert_eq!(summaries[0].category, ToolCategory::Subagent); + assert_eq!(summaries[1].category, ToolCategory::Other); + assert_eq!(summaries[1].description, ""); + assert!(!summaries[0].invoked); + } +} From 1c303326369d630ceb6a506838b1134f36bca8a2 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:05:32 -0400 Subject: [PATCH 08/16] Key the fold's stages by a typed FiringKey instead of a formatted string stage_key formatted ":" and five places re-derived or re-parsed that string: the engine, progress and platform folds, the projector's ordering rule, and fork.rs's stage_labels, which split it back apart. FiringKey is the fact itself, with of_event for the event side and From for the platform-record side. It still serializes as ":", so the fold_json a stored view holds keeps its shape and needs no migration; a unit test pins that. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/fork.rs | 16 +-- .../fabro-petri/src/projection/coordinator.rs | 4 +- .../fabro-petri/src/projection/engine.rs | 19 ++- .../fabro-petri/src/projection/mod.rs | 112 +++++++++++++++--- .../fabro-petri/src/projection/platform.rs | 6 +- .../fabro-petri/src/projection/progress.rs | 14 +-- lib/components/fabro-petri/src/projector.rs | 28 ++--- 7 files changed, 136 insertions(+), 63 deletions(-) diff --git a/lib/components/fabro-petri/src/fork.rs b/lib/components/fabro-petri/src/fork.rs index 70f76dd95..764777029 100644 --- a/lib/components/fabro-petri/src/fork.rs +++ b/lib/components/fabro-petri/src/fork.rs @@ -509,16 +509,12 @@ pub async fn stage_labels(views: &DbPool, run_id: RunId) -> Result) { @@ -59,7 +59,7 @@ impl RunView { } => { self.state .finished_firings - .insert(stage_key(execution.raw(), firing.raw())); + .insert(FiringKey::new(execution.raw(), firing.raw())); let is_final = matches!( event.derived, Some(Derived::StepFinished { is_final: true, .. }) @@ -139,12 +139,11 @@ impl RunView { answer: Some(answer), }) = &event.derived { - let firing_key = event.subject.as_ref().and_then(|subject| { - subject - .firing - .map(|firing| stage_key(execution.raw(), firing.raw())) - }); - self.close_questions(answer.question.as_deref(), firing_key.as_deref(), at); + self.close_questions( + answer.question.as_deref(), + FiringKey::of_event(event), + at, + ); } } // ── Sandbox: the instance (VIEWS.md "Sandbox") ────────────────── @@ -271,7 +270,7 @@ impl RunView { results, .. } => { - let key = stage_key(occurrence.execution.raw(), occurrence.firing.raw()); + let key = FiringKey::new(occurrence.execution.raw(), occurrence.firing.raw()); let Some(stage_id) = self .state .stages @@ -317,7 +316,7 @@ impl RunView { let Some(firing) = subject.firing else { return; }; - let key = stage_key(execution.raw(), firing.raw()); + let key = FiringKey::new(execution.raw(), firing.raw()); if self.state.stages.contains_key(&key) { return; } diff --git a/lib/components/fabro-petri/src/projection/mod.rs b/lib/components/fabro-petri/src/projection/mod.rs index 6cc9f2cbf..e7a4b9cd8 100644 --- a/lib/components/fabro-petri/src/projection/mod.rs +++ b/lib/components/fabro-petri/src/projection/mod.rs @@ -33,15 +33,18 @@ mod progress; mod sandbox; use std::collections::{BTreeMap, BTreeSet}; +use std::fmt; +use std::str::FromStr; use chrono::{DateTime, TimeZone as _, Utc}; +use fabro_store::StagePosition; use fabro_store::platform_records::StoredPlatformRecord; use fabro_types::{ RunControlAction, RunDiff, RunId, RunProjection, RunStatus, StageId, StageProjection, }; use petri_execution::ExecutionId; use petri_execution::events::{NodeRef, RunEvent, Subject}; -use serde::{Deserialize, Serialize}; +use serde::{Deserialize, Deserializer, Serialize, Serializer, de}; use serde_json::Value; use tracing::debug; @@ -91,9 +94,9 @@ pub struct RecordHealth { /// The fold's bookkeeping between items. #[derive(Clone, Debug, Default, Serialize, Deserialize)] pub struct FoldState { - /// Stages by `":"`. + /// Stages by firing. #[serde(default)] - pub stages: BTreeMap, + pub stages: BTreeMap, /// Labels taken, so a second firing with the same name and visit gets /// its own. #[serde(default)] @@ -103,9 +106,9 @@ pub struct FoldState { /// Which invocation each execution belongs to. #[serde(default)] pub executions: BTreeMap, - /// Open questions by id: the stage that asked. + /// Open questions by id: the firing that asked. #[serde(default)] - pub questions: BTreeMap, + pub questions: BTreeMap, #[serde(default, skip_serializing_if = "Option::is_none")] pub root: Option, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -126,11 +129,10 @@ pub struct FoldState { pub run_diff: Option, #[serde(default)] pub health: RecordHealth, - /// Firings (`":"`) whose attempt has recorded a - /// finish: what a position-keyed platform record may be streamed - /// behind. + /// Firings whose attempt has recorded a finish: what a position-keyed + /// platform record may be streamed behind. #[serde(default)] - pub finished_firings: BTreeSet, + pub finished_firings: BTreeSet, /// Whether the run's sandbox still exists after its release /// (`scope.released` `retained`): kept stopped, or deleted. Absent until /// the root invocation's lease was released. The view carries the same @@ -201,7 +203,7 @@ impl RunView { let stage = self .state .stages - .get(&stage_key(execution.raw(), firing.raw()))?; + .get(&FiringKey::new(execution.raw(), firing.raw()))?; if !stage.shown { return None; } @@ -240,10 +242,74 @@ fn settle_control(projection: &mut RunProjection, action: RunControlAction) { } } -/// The key of a stage: its execution and firing. -#[must_use] -pub fn stage_key(execution: u64, firing: u64) -> String { - format!("{execution}:{firing}") +/// The key of a stage: the execution and firing of the visit it shows. The +/// same fact a positioned platform record carries as its `StagePosition`. +/// It is written `:`, which is how the stored fold +/// state keys its maps. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub struct FiringKey { + pub execution: u64, + pub firing: u64, +} + +impl FiringKey { + #[must_use] + pub fn new(execution: u64, firing: u64) -> Self { + Self { execution, firing } + } + + /// The firing an event belongs to: its context's execution and its + /// subject's firing, when it has both. + #[must_use] + pub fn of_event(event: &RunEvent) -> Option { + let execution = event.context.execution?; + let firing = event.subject.as_ref()?.firing?; + Some(Self::new(execution.raw(), firing.raw())) + } +} + +impl From for FiringKey { + fn from(position: StagePosition) -> Self { + Self::new(position.execution, position.firing) + } +} + +impl fmt::Display for FiringKey { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}:{}", self.execution, self.firing) + } +} + +/// A firing key that is not `:`. +#[derive(Debug, thiserror::Error)] +#[error("a firing key is `:`, not {0:?}")] +pub struct ParseFiringKeyError(String); + +impl FromStr for FiringKey { + type Err = ParseFiringKeyError; + + fn from_str(text: &str) -> Result { + let invalid = || ParseFiringKeyError(text.to_string()); + let (execution, firing) = text.split_once(':').ok_or_else(invalid)?; + Ok(Self::new( + execution.parse().map_err(|_| invalid())?, + firing.parse().map_err(|_| invalid())?, + )) + } +} + +impl Serialize for FiringKey { + fn serialize(&self, serializer: S) -> Result { + serializer.collect_str(self) + } +} + +impl<'de> Deserialize<'de> for FiringKey { + fn deserialize>(deserializer: D) -> Result { + String::deserialize(deserializer)? + .parse() + .map_err(de::Error::custom) + } } /// Which firing of its node a subject is, 1-based. @@ -351,4 +417,22 @@ mod tests { assert_eq!(labels, vec!["build@1", "build/e2@1"]); assert_eq!(view.state.stages.len(), 2); } + + #[test] + fn a_firing_key_is_stored_as_execution_colon_firing() { + let mut stages: BTreeMap = BTreeMap::new(); + stages.insert(FiringKey::new(3, 7), 1); + let json = serde_json::to_string(&stages).expect("the map encodes"); + assert_eq!(json, r#"{"3:7":1}"#); + let back: BTreeMap = serde_json::from_str(&json).expect("the map decodes"); + assert_eq!(back, stages); + assert!("3-7".parse::().is_err()); + assert_eq!( + FiringKey::from(StagePosition { + execution: 3, + firing: 7, + }), + FiringKey::new(3, 7) + ); + } } diff --git a/lib/components/fabro-petri/src/projection/platform.rs b/lib/components/fabro-petri/src/projection/platform.rs index ebff0cd24..7109e25fd 100644 --- a/lib/components/fabro-petri/src/projection/platform.rs +++ b/lib/components/fabro-petri/src/projection/platform.rs @@ -17,7 +17,7 @@ use fabro_types::{ use tracing::debug; use super::sandbox::sandbox_plan; -use super::{RunView, apply_status, millis, settle_control, stage_key, touch}; +use super::{FiringKey, RunView, apply_status, millis, settle_control, touch}; impl RunView { pub(super) fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { @@ -76,7 +76,7 @@ impl RunView { let stage = self .state .stages - .get(&stage_key(record.execution, record.firing)); + .get(&FiringKey::new(record.execution, record.firing)); let current_node = stage.map_or_else(String::new, |stage| stage.node_name.clone()); let stage_id = stage .filter(|stage| stage.shown) @@ -107,7 +107,7 @@ impl RunView { let stage = self .state .stages - .get(&stage_key(record.execution, record.firing)); + .get(&FiringKey::new(record.execution, record.firing)); let Some(stage_id) = stage.map(|stage| stage.stage_id.clone()) else { debug!( seq = stored.seq, diff --git a/lib/components/fabro-petri/src/projection/progress.rs b/lib/components/fabro-petri/src/projection/progress.rs index 09121fac1..f5a0eac25 100644 --- a/lib/components/fabro-petri/src/projection/progress.rs +++ b/lib/components/fabro-petri/src/projection/progress.rs @@ -20,7 +20,7 @@ use serde_json::Value; use tracing::debug; use super::model::{model_ref, split_model}; -use super::{RunView, apply_status, stage_key}; +use super::{FiringKey, RunView, apply_status}; use crate::interview::question_type; impl RunView { @@ -115,7 +115,7 @@ impl RunView { let group = self .state .stages - .get(&stage_key(execution.raw(), occurrence.firing)) + .get(&FiringKey::new(execution.raw(), occurrence.firing)) .map(|stage| stage.stage_id.clone()); if let Some(group) = group { self.state.invocations.entry(invocation).or_default().branch = @@ -143,7 +143,7 @@ impl RunView { let Some(firing) = subject.firing else { return; }; - let key = stage_key(execution.raw(), firing.raw()); + let key = FiringKey::new(execution.raw(), firing.raw()); let label = self.state.stages.get(&key).map_or_else( || subject.node.name.to_string(), |stage| stage.stage_id.to_string(), @@ -201,16 +201,16 @@ impl RunView { pub(super) fn close_questions( &mut self, question: Option<&str>, - firing_key: Option<&str>, + firing: Option, at: DateTime, ) { - let closed: Vec = match (question, firing_key) { + let closed: Vec = match (question, firing) { (Some(question), _) => vec![question.to_string()], - (None, Some(key)) => self + (None, Some(firing)) => self .state .questions .iter() - .filter(|(_, asked_by)| asked_by.as_str() == key) + .filter(|(_, asked_by)| **asked_by == firing) .map(|(id, _)| id.clone()) .collect(), (None, None) => Vec::new(), diff --git a/lib/components/fabro-petri/src/projector.rs b/lib/components/fabro-petri/src/projector.rs index b9b3e274d..533a590c5 100644 --- a/lib/components/fabro-petri/src/projector.rs +++ b/lib/components/fabro-petri/src/projector.rs @@ -77,7 +77,7 @@ use tracing::{debug, info, warn}; use self::cache::{Caches, IDLE, RunCache}; use crate::SqliteRunStore; -use crate::projection::{self, FoldState, Item, RecordHealth, RunView}; +use crate::projection::{self, FiringKey, FoldState, Item, RecordHealth, RunView}; /// The positions a view committed: the last event consumed per Petri log, /// and the last platform record consumed. @@ -891,23 +891,16 @@ impl petri_execution::RunLogs for SignallingLogs { fn order_items<'a>( events: &'a [RunEvent], platform_records: &'a [StoredPlatformRecord], - finished_before: &BTreeSet, + finished_before: &BTreeSet, run_finished: bool, ) -> (Vec>, usize) { - let firing_of = |event: &RunEvent| -> Option<(u64, u64)> { - let execution = event.context.execution?; - let firing = event.subject.as_ref()?.firing?; - Some((execution.raw(), firing.raw())) - }; - let finished_in_pass = |at: (u64, u64)| { + let finished_in_pass = |at: FiringKey| { events.iter().any(|event| { - firing_of(event) == Some(at) + FiringKey::of_event(event) == Some(at) && matches!(event.engine(), Some(Event::StepFinished { .. })) }) }; - let finished = |at: (u64, u64)| { - finished_in_pass(at) || finished_before.contains(&projection::stage_key(at.0, at.1)) - }; + let finished = |at: FiringKey| finished_in_pass(at) || finished_before.contains(&at); // Platform records are consumed in seq order: the first one whose firing // has not finished holds itself and everything after it. let consumed = if run_finished { @@ -918,7 +911,7 @@ fn order_items<'a>( .position(|record| { record .position - .is_some_and(|position| !finished((position.execution, position.firing))) + .is_some_and(|position| !finished(FiringKey::from(position))) }) .unwrap_or(platform_records.len()) }; @@ -940,7 +933,7 @@ fn order_items<'a>( items.sort_by_key(|(recorded_at, rank, _)| (*recorded_at, *rank)); let item_firing = |item: &Item<'a>| match item { - Item::Petri(event) => firing_of(event), + Item::Petri(event) => FiringKey::of_event(event), Item::Platform(_) => None, }; let is_routing = |item: &Item<'a>| { @@ -959,7 +952,7 @@ fn order_items<'a>( let Some(position) = record.position else { continue; }; - let at = (position.execution, position.firing); + let at = FiringKey::from(position); let first_routing = items .iter() .position(|(_, _, other)| item_firing(other) == Some(at) && is_routing(other)); @@ -967,7 +960,8 @@ fn order_items<'a>( .iter() .rposition(|(_, _, other)| item_firing(other) == Some(at)); let first_later = items.iter().position(|(_, _, other)| { - item_firing(other).is_some_and(|(execution, firing)| execution == at.0 && firing > at.1) + item_firing(other) + .is_some_and(|key| key.execution == at.execution && key.firing > at.firing) }); keys[index] = if let Some(before) = first_routing { (before, 0) @@ -1323,7 +1317,7 @@ mod tests { fn a_positioned_record_precedes_later_firings_and_an_unpositioned_one_keeps_its_clock() { let events = vec![started(13, 2, 103), finished(14, 2, 104)]; let records = vec![checkpoint(1, 1, 250)]; - let finished_before: BTreeSet = [projection::stage_key(0, 1)].into_iter().collect(); + let finished_before: BTreeSet = [FiringKey::new(0, 1)].into_iter().collect(); let (items, held) = order_items(&events, &records, &finished_before, false); assert_eq!(held, 0); assert_eq!(names(&items), vec![ From 4216364e92527221a33497189183039eaf859674 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:10:03 -0400 Subject: [PATCH 09/16] Split the projector into the pass, its ordering rule, its stream and its wake-up projector.rs mixed the commit rule with things that are not the pass: the ordering rule and its tests, the in-process signalling store, the stream table's rows and reads, and six readers documented "for a test". projector/mod.rs now holds the pass alone; order.rs, signalling.rs and stream.rs hold the rest; the test readers (rebuild, stored_projection, stored_stream, stored_platform_records, event_json) live in test_support behind the test-support feature, as the test-support boundary rule asks, and the two test callers reach them there. The fault injection and the cache's test-only reader are gated the same way, so neither ships. pass() reads as three steps: at_head, read_new_events (a NewEvents struct in place of a tuple) and stream::stream_rows, which rebuild shares instead of repeating the fold loop. PassReport::skipped and PassReport::contended replace the zero-filled literals. Co-Authored-By: Claude Fable 5.1 --- .../fabro-server/tests/it/scenario/petri.rs | 4 +- .../fabro-petri/src/projector/cache.rs | 1 + .../src/{projector.rs => projector/mod.rs} | 850 ++++-------------- .../fabro-petri/src/projector/order.rs | 343 +++++++ .../fabro-petri/src/projector/signalling.rs | 99 ++ .../fabro-petri/src/projector/stream.rs | 113 +++ .../fabro-petri/src/test_support.rs | 120 ++- .../fabro-petri/tests/projection.rs | 71 +- 8 files changed, 869 insertions(+), 732 deletions(-) rename lib/components/fabro-petri/src/{projector.rs => projector/mod.rs} (52%) create mode 100644 lib/components/fabro-petri/src/projector/order.rs create mode 100644 lib/components/fabro-petri/src/projector/signalling.rs create mode 100644 lib/components/fabro-petri/src/projector/stream.rs diff --git a/lib/apps/fabro-server/tests/it/scenario/petri.rs b/lib/apps/fabro-server/tests/it/scenario/petri.rs index 9e53c3381..9ad651660 100644 --- a/lib/apps/fabro-server/tests/it/scenario/petri.rs +++ b/lib/apps/fabro-server/tests/it/scenario/petri.rs @@ -28,7 +28,7 @@ use axum::body::Body; use axum::http::{Request, StatusCode}; use fabro_petri::engine::{self, RunStatus}; use fabro_petri::petri::{Access, OwnerId, RunKey, RunStore as _}; -use fabro_petri::{SqliteRunStore, projector}; +use fabro_petri::{SqliteRunStore, test_support}; use fabro_server::server::AppState; use fabro_server::test_support::{ TestAppStateBuilder, llm_overlay_with_provider_base_url, test_app_db_pool, @@ -237,7 +237,7 @@ pub(super) async fn settled_state( /// How many items the run's projected stream holds. async fn petri_stream_len(state: &AppState, run_id: &str) -> usize { let id: RunId = run_id.parse().expect("the run id parses"); - projector::stored_stream(&state.test_petri_view_pool(), id) + test_support::stored_stream(&state.test_petri_view_pool(), id) .await .expect("the stream reads") .len() diff --git a/lib/components/fabro-petri/src/projector/cache.rs b/lib/components/fabro-petri/src/projector/cache.rs index c2f423443..abe4cf5fd 100644 --- a/lib/components/fabro-petri/src/projector/cache.rs +++ b/lib/components/fabro-petri/src/projector/cache.rs @@ -121,6 +121,7 @@ impl Caches { } /// Whether a cache is kept for the run: a test's view of the cache. + #[cfg(any(test, feature = "test-support"))] pub(crate) fn holds(&self, run_id: RunId) -> bool { let runs = sync::lock(&self.runs); runs.get(&run_id) diff --git a/lib/components/fabro-petri/src/projector.rs b/lib/components/fabro-petri/src/projector/mod.rs similarity index 52% rename from lib/components/fabro-petri/src/projector.rs rename to lib/components/fabro-petri/src/projector/mod.rs index 533a590c5..cada3395c 100644 --- a/lib/components/fabro-petri/src/projector.rs +++ b/lib/components/fabro-petri/src/projector/mod.rs @@ -54,21 +54,24 @@ //! completeness once the run has recorded its finish. mod cache; +pub(crate) mod order; +mod signalling; +pub(crate) mod stream; -use std::collections::{BTreeMap, BTreeSet, HashMap}; +use std::collections::{BTreeMap, HashMap}; +#[cfg(any(test, feature = "test-support"))] use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex}; use std::time::Duration; use fabro_db::DbPool; -use fabro_store::platform_records::{PlatformRecordStore, StoredPlatformRecord, now_ms}; +use fabro_store::platform_records::{PlatformRecordStore, now_ms}; use fabro_store::{RunProjection, RunSummaryStore}; -use fabro_types::{RunId, RunStreamItem, RunStreamItemKind}; +use fabro_types::{RunId, RunStreamItem}; use fabro_util::error::collect_chain; use fabro_util::sync; -use petri_execution::events::{self, EventId, EventSource, RunEvent}; +use petri_execution::events::{EventId, EventSource, RunEvent}; use petri_execution::{Access, CoordinatorEvent, RunKey, RunStore as _, inspect}; -use petri_runtime::engine::Event; use petri_store::StoreError; use serde::{Deserialize, Serialize}; use tokio::sync::broadcast; @@ -76,8 +79,9 @@ use tokio::time; use tracing::{debug, info, warn}; use self::cache::{Caches, IDLE, RunCache}; +use self::stream::StreamRow; use crate::SqliteRunStore; -use crate::projection::{self, FiringKey, FoldState, Item, RecordHealth, RunView}; +use crate::projection::{self, FoldState, RecordHealth, RunView}; /// The positions a view committed: the last event consumed per Petri log, /// and the last platform record consumed. @@ -128,6 +132,48 @@ pub struct PassReport { pub health: RecordHealth, } +impl PassReport { + /// A pass that found the view at the head of every log and wrote + /// nothing. + fn skipped(run_id: RunId, stored: &StoredView) -> Self { + Self { + run_id, + skipped: true, + contended: false, + petri_events: 0, + platform_records: 0, + replayed_records: 0, + stream_seq: stored.stream_seq, + positions: stored.positions.clone(), + health: stored.view.state.health.clone(), + } + } + + /// A pass that left the view alone because a platform record landed + /// under it; the projector runs it again. + fn contended(run_id: RunId, replayed_records: usize) -> Self { + Self { + run_id, + skipped: false, + contended: true, + petri_events: 0, + platform_records: 0, + replayed_records, + stream_seq: 0, + positions: Positions::default(), + health: RecordHealth::default(), + } + } +} + +/// What a pass read past its cache: the events new to the view, the +/// replay failure that held it, and how many records the replay cost. +struct NewEvents { + events: Vec, + replay_failure: Option, + replayed_records: usize, +} + /// What the startup pass did. #[derive(Clone, Debug, Default, PartialEq, Eq)] pub struct StartupReport { @@ -182,8 +228,9 @@ pub struct Projector { /// over the same run never interleave their reads and writes), and the /// cache each live run's passes continue from. pub(crate) caches: Caches, - /// Test-only: stop the next pass after its reads, before its view - /// transaction, as a crash there would. + /// A test's fault: stop the next pass after its reads, before its + /// view transaction, as a crash there would. + #[cfg(any(test, feature = "test-support"))] fault: AtomicBool, /// Sent after each committed pass that wrote stream rows: the run whose /// stream grew. A wake-up for the stream's readers, never a source of @@ -210,6 +257,7 @@ impl Projector { pool: views, slots: Mutex::default(), caches: Caches::default(), + #[cfg(any(test, feature = "test-support"))] fault: AtomicBool::new(false), committed: broadcast::channel(COMMIT_SIGNAL_CAPACITY).0, }) @@ -232,7 +280,7 @@ impl Projector { after: u64, limit: usize, ) -> Result, ProjectError> { - stream_after(&self.pool, run_id, after, limit).await + stream::stream_after(&self.pool, run_id, after, limit).await } /// Delete everything the store and the view tables hold for the run: @@ -344,10 +392,24 @@ impl Projector { /// Stop the next pass after its reads and before its view transaction, /// as a crash there would, once. + #[cfg(any(test, feature = "test-support"))] pub fn fail_before_view(&self) { self.fault.store(true, Ordering::SeqCst); } + /// Whether a test asked this pass to stop before its view transaction; + /// never outside tests. + fn take_fault(&self) -> bool { + #[cfg(any(test, feature = "test-support"))] + { + self.fault.swap(false, Ordering::SeqCst) + } + #[cfg(not(any(test, feature = "test-support")))] + { + false + } + } + /// One pass over every Petri run the database holds: the runs with a /// Petri record, and the runs with platform records. Runs whose view /// already covers every committed record are skipped cheaply. @@ -430,67 +492,20 @@ impl Projector { /// positions the cache's view holds, fold it, and write the view. async fn pass(&self, run_id: RunId, run: &mut RunCache) -> Result { let key = RunKey::new(run_id.to_string()); - let platform_head = self - .platform - .head(&run_id) - .await - .map_err(ProjectError::Store)? - .unwrap_or(0); - let petri_heads = self.petri_heads(&run_id).await?; - let stored = &run.view; - let at_head = platform_head == stored.positions.platform_seq - && petri_heads.iter().all(|(log, head)| { - stored - .positions - .petri - .iter() - .any(|held| log_text(&held.source) == *log && held.seq == *head) - }); - if at_head && stored.view.projection.is_some() { - return Ok(PassReport { - run_id, - skipped: true, - contended: false, - petri_events: 0, - platform_records: 0, - replayed_records: 0, - stream_seq: stored.stream_seq, - positions: stored.positions.clone(), - health: stored.view.state.health.clone(), - }); + if self.at_head(run_id, &run.view).await? { + return Ok(PassReport::skipped(run_id, &run.view)); } let platform_records = self .platform - .read_after(&run_id, stored.positions.platform_seq) + .read_after(&run_id, run.view.positions.platform_seq) .await .map_err(ProjectError::Store)?; - let mut replayed_records = 0; - let (events, replay_failure) = match self.store.open(&key, Access::Read).await { - Ok(logs) => match run.replay.advance(&*logs).await { - Ok(new) => { - replayed_records = new.iter().filter(|event| event.id.index == 0).count(); - // A rebuilt replay derives the run whole: only the events - // past the view's positions are new to it. - let held = run.view.positions.held(); - let mut events = std::mem::take(&mut run.pending); - events.extend(new.into_iter().filter(|event| { - held.get(&event.id.source) - .is_none_or(|last| event.id > *last) - })); - (events, None) - } - // The replay stood still and is retried by the next pass; - // what it derived before stays pending. - Err(error) => { - let chain = collect_chain(&error).join(": "); - warn!(run_id = %run_id, error = %chain, "Petri run does not replay; the view holds"); - (Vec::new(), Some(chain)) - } - }, - Err(StoreError::NotFound { .. }) => (Vec::new(), None), - Err(error) => return Err(ProjectError::Open(error)), - }; + let NewEvents { + events, + replay_failure, + replayed_records, + } = self.read_new_events(run_id, &key, run).await?; let mut view = run.view.view.clone(); let mut positions = run.view.positions.clone(); @@ -505,7 +520,7 @@ impl Projector { let platform_head_seen = platform_records .last() .map_or(positions.platform_seq, |record| record.seq); - let (items, held) = order_items( + let (items, held) = order::order_items( &events, &platform_records, &view.state.finished_firings, @@ -518,37 +533,11 @@ impl Projector { "platform records held back until their firing's finish is in the stream" ); } - - let mut rows: Vec = Vec::with_capacity(items.len()); - for item in &items { - stream_seq += 1; - view.fold(item, stream_seq); - let row = match item { - Item::Petri(event) => { - positions.advance(event.id); - StreamRow { - stream_seq, - item_kind: "petri", - item_id: event_id_text(&event.id), - event_json: serde_json::to_string(event).map_err(ProjectError::Encode)?, - } - } - Item::Platform(record) => { - positions.platform_seq = record.seq; - StreamRow { - stream_seq, - item_kind: "platform", - item_id: record.seq.to_string(), - event_json: serde_json::to_string(record).map_err(ProjectError::Encode)?, - } - } - }; - rows.push(row); - } + let rows = stream::stream_rows(&items, &mut view, &mut positions, &mut stream_seq)?; drop(items); view.state.health = self.health(&key, &view.state, replay_failure).await?; - if self.fault.swap(false, Ordering::SeqCst) { + if self.take_fault() { run.pending = events; return Err(ProjectError::Injected); } @@ -567,17 +556,7 @@ impl Projector { Ok(false) => { debug!(run_id = %run_id, "platform records landed during the pass; running it again"); run.pending = events; - return Ok(PassReport { - run_id, - skipped: false, - contended: true, - petri_events: 0, - platform_records: 0, - replayed_records, - stream_seq: 0, - positions: Positions::default(), - health: RecordHealth::default(), - }); + return Ok(PassReport::contended(run_id, replayed_records)); } Err(error) => { run.pending = events; @@ -612,6 +591,77 @@ impl Projector { }) } + /// Whether the stored view already covers every committed record: the + /// platform head and every Petri log's head are the positions it holds, + /// and it has a projection to serve. + async fn at_head(&self, run_id: RunId, stored: &StoredView) -> Result { + let platform_head = self + .platform + .head(&run_id) + .await + .map_err(ProjectError::Store)? + .unwrap_or(0); + let petri_heads = self.petri_heads(&run_id).await?; + Ok(platform_head == stored.positions.platform_seq + && petri_heads.iter().all(|(log, head)| { + stored + .positions + .petri + .iter() + .any(|held| stream::log_text(&held.source) == *log && held.seq == *head) + }) + && stored.view.projection.is_some()) + } + + /// The events past the view's positions: the cache's replay advanced + /// over the records committed since, led by the events an earlier pass + /// derived and did not commit. A rebuilt replay derives the run whole, + /// so only the events past the view's positions are new to it. A + /// replay that fails (a torn tail) stands still, yields nothing, and + /// names its reason; what it derived before stays pending for the next + /// pass. + async fn read_new_events( + &self, + run_id: RunId, + key: &RunKey, + run: &mut RunCache, + ) -> Result { + let nothing = NewEvents { + events: Vec::new(), + replay_failure: None, + replayed_records: 0, + }; + let logs = match self.store.open(key, Access::Read).await { + Ok(logs) => logs, + Err(StoreError::NotFound { .. }) => return Ok(nothing), + Err(error) => return Err(ProjectError::Open(error)), + }; + match run.replay.advance(&*logs).await { + Ok(new) => { + let replayed_records = new.iter().filter(|event| event.id.index == 0).count(); + let held = run.view.positions.held(); + let mut events = std::mem::take(&mut run.pending); + events.extend(new.into_iter().filter(|event| { + held.get(&event.id.source) + .is_none_or(|last| event.id > *last) + })); + Ok(NewEvents { + events, + replay_failure: None, + replayed_records, + }) + } + Err(error) => { + let chain = collect_chain(&error).join(": "); + warn!(run_id = %run_id, error = %chain, "Petri run does not replay; the view holds"); + Ok(NewEvents { + replay_failure: Some(chain), + ..nothing + }) + } + } + } + /// The view transaction: the projection row, the stream rows and the /// `runs` row, committed together, unless a platform record landed /// since the pass read them (`false`: the view is left alone). @@ -655,8 +705,8 @@ impl Projector { .bind(projection_json) .bind(fold_json) .bind(positions_json) - .bind(column(stream_seq)) - .bind(column(now_ms())) + .bind(stream::column(stream_seq)) + .bind(stream::column(now_ms())) .execute(&mut *tx) .await .map_err(ProjectError::Database)?; @@ -666,7 +716,7 @@ impl Projector { VALUES (?, ?, ?, ?, ?)", ) .bind(run_id.to_string()) - .bind(column(row.stream_seq)) + .bind(stream::column(row.stream_seq)) .bind(row.item_kind) .bind(&row.item_id) .bind(&row.event_json) @@ -772,335 +822,13 @@ impl Projector { } } -impl Projector { - /// A run store whose appends signal this projector: for a run that - /// executes in the same process as the projector, over the SQLite store - /// directly, where no append endpoint is there to signal. The signal is - /// sent after the store's append returned, so the records it covers are - /// durable before the view sees them. - pub fn observe_store( - self: &Arc, - inner: Arc, - ) -> Arc { - Arc::new(SignallingStore { - inner, - projector: Arc::clone(self), - }) - } -} - -/// A run store that signals a projector after each append. -struct SignallingStore { - inner: Arc, - projector: Arc, -} - -#[async_trait::async_trait] -impl petri_execution::RunStore for SignallingStore { - async fn open( - &self, - key: &RunKey, - access: Access, - ) -> Result, StoreError> { - let logs = self.inner.open(key, access).await?; - Ok(Arc::new(SignallingLogs { - inner: logs, - run_id: projection::run_id_of(key.as_str()), - projector: Arc::clone(&self.projector), - })) - } -} - -struct SignallingLogs { - inner: Arc, - run_id: Option, - projector: Arc, -} - -#[async_trait::async_trait] -impl petri_execution::RunLogs for SignallingLogs { - fn locator(&self) -> String { - self.inner.locator() - } - - async fn append( - &self, - log: &petri_execution::LogId, - records: &[petri_execution::Record], - ) -> Result<(), StoreError> { - self.inner.append(log, records).await?; - if let Some(run_id) = self.run_id { - self.projector.signal(run_id); - } - Ok(()) - } - - async fn read( - &self, - log: &petri_execution::LogId, - ) -> Result, StoreError> { - self.inner.read(log).await - } - - async fn read_from( - &self, - log: &petri_execution::LogId, - seq: u64, - ) -> Result, StoreError> { - self.inner.read_from(log, seq).await - } - - async fn put_blob(&self, bytes: &[u8]) -> Result { - self.inner.put_blob(bytes).await - } - - async fn get_blob(&self, digest: petri_store::Digest) -> Result>, StoreError> { - self.inner.get_blob(digest).await - } -} - -/// The order one pass streams its new items in, and how many platform -/// records it holds back for a later pass. -/// -/// Every item is first ordered by `recorded_at` (stable: the coordinator -/// log before an execution log before a platform record on a tie, and each -/// log's own order kept). A platform record that carries a Petri position -/// (a checkpoint, keyed on `(execution, firing)`) is then placed by that -/// position, not by its clock, because the server stamps the record and the -/// worker stamps Petri's records and the two clocks can tie or invert: -/// -/// - before the firing's first `routing.resolved` event in the pass, which is -/// right after the firing's finish (its `step.finished` and the -/// `visit.completed` attached to it) and before the next firing's -/// `visit.started`, which is attached to that routing record; -/// - else after the last event of the firing in the pass; -/// - else, when the firing finished in an earlier pass, before the first event -/// of a later firing (a larger firing id) in the same execution, or where its -/// `recorded_at` put it; -/// - else the record is held back, with every platform record after it, and the -/// pass consumes platform records only up to it. The hook that writes a -/// checkpoint record runs after the driver appended the attempt's finish, but -/// the driver's store writer flushes that record on its own schedule, so the -/// platform record can be committed before its firing's `step.finished`; -/// holding it keeps the stream's order the same live and on a rebuild. -/// Nothing is held once the run has recorded its finish. -/// -/// The rule reads only the pass's own items and the firings already -/// finished, so a record is never streamed before its firing's finish and -/// never after the firing's routes. -fn order_items<'a>( - events: &'a [RunEvent], - platform_records: &'a [StoredPlatformRecord], - finished_before: &BTreeSet, - run_finished: bool, -) -> (Vec>, usize) { - let finished_in_pass = |at: FiringKey| { - events.iter().any(|event| { - FiringKey::of_event(event) == Some(at) - && matches!(event.engine(), Some(Event::StepFinished { .. })) - }) - }; - let finished = |at: FiringKey| finished_in_pass(at) || finished_before.contains(&at); - // Platform records are consumed in seq order: the first one whose firing - // has not finished holds itself and everything after it. - let consumed = if run_finished { - platform_records.len() - } else { - platform_records - .iter() - .position(|record| { - record - .position - .is_some_and(|position| !finished(FiringKey::from(position))) - }) - .unwrap_or(platform_records.len()) - }; - let held = platform_records.len() - consumed; - let platform_records = &platform_records[..consumed]; - - let mut items: Vec<(u64, u8, Item<'a>)> = - Vec::with_capacity(events.len() + platform_records.len()); - for event in events { - let rank = match event.id.source { - EventSource::Coordinator => 0, - EventSource::Execution { .. } => 1, - }; - items.push((event.recorded_at, rank, Item::Petri(event))); - } - for record in platform_records { - items.push((record.recorded_at, 2, Item::Platform(record))); - } - items.sort_by_key(|(recorded_at, rank, _)| (*recorded_at, *rank)); - - let item_firing = |item: &Item<'a>| match item { - Item::Petri(event) => FiringKey::of_event(event), - Item::Platform(_) => None, - }; - let is_routing = |item: &Item<'a>| { - matches!( - item, - Item::Petri(event) if matches!(event.engine(), Some(Event::RoutingResolved { .. })) - ) - }; - // The key of each item: its index in clock order, and whether it sits - // before (0), at (1) or after (2) that index. - let mut keys: Vec<(usize, u8)> = (0..items.len()).map(|index| (index, 1)).collect(); - for (index, (_, _, item)) in items.iter().enumerate() { - let Item::Platform(record) = item else { - continue; - }; - let Some(position) = record.position else { - continue; - }; - let at = FiringKey::from(position); - let first_routing = items - .iter() - .position(|(_, _, other)| item_firing(other) == Some(at) && is_routing(other)); - let last_of_firing = items - .iter() - .rposition(|(_, _, other)| item_firing(other) == Some(at)); - let first_later = items.iter().position(|(_, _, other)| { - item_firing(other) - .is_some_and(|key| key.execution == at.execution && key.firing > at.firing) - }); - keys[index] = if let Some(before) = first_routing { - (before, 0) - } else if let Some(after) = last_of_firing { - (after, 2) - } else if let Some(before) = first_later { - (before, 0) - } else { - (index, 1) - }; - } - let mut order: Vec = (0..items.len()).collect(); - order.sort_by_key(|index| keys[*index]); - let mut ordered: Vec>> = - items.into_iter().map(|(_, _, item)| Some(item)).collect(); - let items = order - .into_iter() - .map(|index| ordered[index].take().expect("each item is placed once")) - .collect(); - (items, held) -} - -struct StreamRow { - stream_seq: u64, - item_kind: &'static str, - item_id: String, - event_json: String, -} - /// How many commit signals a slow reader may fall behind before it is told /// it lagged and re-reads from its cursor. const COMMIT_SIGNAL_CAPACITY: usize = 1024; -/// The run's stream past the cursor, read from the view tables: up to -/// `limit` rows with `stream_seq > after`, in order, in Fabro's envelope. -pub async fn stream_after( - views: &DbPool, - run_id: RunId, - after: u64, - limit: usize, -) -> Result, ProjectError> { - let rows: Vec<(i64, String, String, String)> = sqlx::query_as( - "SELECT stream_seq, item_kind, item_id, event_json FROM petri_stream WHERE run_id = ? AND \ - stream_seq > ? ORDER BY stream_seq LIMIT ?", - ) - .bind(run_id.to_string()) - .bind(column(after)) - .bind(i64::try_from(limit).unwrap_or(i64::MAX)) - .fetch_all(views) - .await - .map_err(ProjectError::Database)?; - rows.into_iter() - .map(|(stream_seq, item_kind, item_id, event_json)| { - let item: serde_json::Value = - serde_json::from_str(&event_json).map_err(ProjectError::Encode)?; - let kind = match item_kind.as_str() { - "platform" => RunStreamItemKind::Platform, - _ => RunStreamItemKind::Petri, - }; - let recorded_at = item - .get("recorded_at") - .and_then(serde_json::Value::as_u64) - .unwrap_or(0); - Ok(RunStreamItem { - run_id, - stream_seq: u64::try_from(stream_seq).unwrap_or(0), - kind, - id: item_id, - recorded_at, - item, - }) - }) - .collect() -} - -/// A Petri event id as the stream names it: `//`. -#[must_use] -pub fn event_id_text(id: &EventId) -> String { - format!("{}/{}/{}", log_text(&id.source), id.seq, id.index) -} - -fn log_text(source: &EventSource) -> String { - match source { - EventSource::Coordinator => "coordinator".to_string(), - EventSource::Execution { execution } => format!("execution {execution}"), - } -} - -fn column(value: u64) -> i64 { - i64::try_from(value).unwrap_or(i64::MAX) -} - -/// The run's projection rebuilt from its records alone, with nothing -/// stored: what a fresh projector would commit over the same records. A test -/// compares it with the live view. `records` and `views` are the two pools -/// [`Projector::new`] takes. -pub async fn rebuild( - records: &DbPool, - views: &DbPool, - run_id: RunId, -) -> Result<(Option, Positions, u64), ProjectError> { - let store = SqliteRunStore::new(records.clone()); - let platform = PlatformRecordStore::new(views.clone()); - let key = RunKey::new(run_id.to_string()); - let platform_records = platform.read(&run_id).await.map_err(ProjectError::Store)?; - let events = match store.open(&key, Access::Read).await { - Ok(logs) => events::replay_run(&*logs) - .await - .inspect_err(|error| { - warn!(error = %collect_chain(error).join(": "), "rebuild: the run does not replay"); - }) - .unwrap_or_default(), - Err(StoreError::NotFound { .. }) => Vec::new(), - Err(error) => return Err(ProjectError::Open(error)), - }; - let run_finished = events.iter().any(|event| { - matches!( - event.coordinator(), - Some(CoordinatorEvent::RunFinished { .. }) - ) - }); - let (items, _held) = order_items(&events, &platform_records, &BTreeSet::new(), run_finished); - let mut view = RunView::new(); - let mut positions = Positions::default(); - let mut stream_seq = 0; - for item in &items { - stream_seq += 1; - view.fold(item, stream_seq); - match item { - Item::Petri(event) => positions.advance(event.id), - Item::Platform(record) => positions.platform_seq = record.seq, - } - } - Ok((view.projection, positions, stream_seq)) -} - -/// The stored view's positions and stream sequence, for a test; `views` is -/// the pool the view tables live in. -pub async fn stored_positions( +/// The stored view's positions and stream sequence; `views` is the pool the +/// view tables live in. +pub(crate) async fn stored_positions( views: &DbPool, run_id: RunId, ) -> Result, ProjectError> { @@ -1118,261 +846,3 @@ pub async fn stored_positions( }) .transpose() } - -/// The stored view's projection, for a test or a reader outside the store. -pub async fn stored_projection( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - let json: Option = - sqlx::query_scalar("SELECT projection_json FROM petri_projection WHERE run_id = ?") - .bind(run_id.to_string()) - .fetch_optional(views) - .await - .map_err(ProjectError::Database)?; - json.map(|json| serde_json::from_str(&json).map_err(ProjectError::Encode)) - .transpose() -} - -/// The stream rows of a run: `(stream_seq, item_kind, item_id)`, in order. -pub async fn stored_stream( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - let rows: Vec<(i64, String, String)> = sqlx::query_as( - "SELECT stream_seq, item_kind, item_id FROM petri_stream WHERE run_id = ? ORDER BY stream_seq", - ) - .bind(run_id.to_string()) - .fetch_all(views) - .await - .map_err(ProjectError::Database)?; - Ok(rows - .into_iter() - .map(|(seq, kind, id)| (u64::try_from(seq).unwrap_or(0), kind, id)) - .collect()) -} - -/// Every stored platform record of a run, for a reader outside the store. -pub async fn stored_platform_records( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - PlatformRecordStore::new(views.clone()) - .read(&run_id) - .await - .map_err(ProjectError::Store) -} - -/// A recorded event's projection is what `RunEvent` serializes to. -#[must_use] -pub fn event_json(event: &RunEvent) -> serde_json::Value { - serde_json::to_value(event).unwrap_or_default() -} - -#[cfg(test)] -mod tests { - use fabro_store::PlatformRecord; - use fabro_store::platform_records::{CheckpointRecord, StagePosition}; - use petri_execution::events::{Context, NodeRef, Record, RecordOrigin, Subject}; - use petri_execution::{ExecutionId, StoredEngineRecord}; - use petri_runtime::driver::BranchRole; - use petri_runtime::engine::{DecisionId, EventOrigin, RouteApplied}; - use petri_runtime::ir::{Attempt, FiringId, NodeId, Outcome, Status}; - - use super::*; - - /// A firing's engine event at `seq`, recorded at `at`. - fn engine_event(seq: u64, firing: u64, at: u64, body: Event) -> RunEvent { - RunEvent { - id: EventId { - source: EventSource::Execution { - execution: ExecutionId::new(0), - }, - seq, - index: 0, - }, - origin: RecordOrigin::External, - context: Context { - invocation: None, - execution: Some(ExecutionId::new(0)), - parent: None, - }, - subject: Some(Subject { - node: NodeRef { - id: NodeId::new(1), - name: format!("n{firing}").into(), - kind: "attractor/command".into(), - meta: serde_json::Value::Null, - }, - firing: Some(FiringId::new(firing)), - visit: Some(1), - attempt: Some(Attempt::FIRST), - generation: None, - branch: BranchRole::None, - }), - observed_at: None, - recorded_at: at, - record: Some(Record::Engine(StoredEngineRecord { - seq, - origin: EventOrigin::External, - recorded_at: at, - body, - })), - derived: None, - } - } - - fn finished(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::StepFinished { - firing: FiringId::new(firing), - attempt: Attempt::FIRST, - outcome: Outcome::new(Status::Success, serde_json::Value::Null), - }) - } - - fn routing(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::RoutingResolved { - decision_id: DecisionId::route(FiringId::new(firing), Attempt::FIRST), - groups: Vec::new(), - }) - } - - fn applied(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::RouteApplied { - applied: RouteApplied::None { - firing: FiringId::new(firing), - group: 0, - }, - }) - } - - fn started(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::StepStarted { - firing: FiringId::new(firing), - attempt: Attempt::FIRST, - }) - } - - fn checkpoint(seq: u64, firing: u64, at: u64) -> StoredPlatformRecord { - StoredPlatformRecord { - seq, - recorded_at: at, - record: PlatformRecord::Checkpoint(CheckpointRecord { - execution: 0, - firing, - attempt: Some(1), - workspace: None, - git_commit_sha: Some("abc".to_string()), - diff_summary: None, - patch_blob: None, - operation: None, - }), - position: Some(StagePosition { - execution: 0, - firing, - }), - } - } - - fn names(items: &[Item<'_>]) -> Vec { - items - .iter() - .map(|item| match item { - Item::Petri(event) => event_id_text(&event.id), - Item::Platform(record) => format!("platform {}", record.seq), - }) - .collect() - } - - /// A firing's events, then a later firing's events, then a checkpoint - /// for the first firing stamped later than all of them: the stream puts - /// the checkpoint right after the first firing's finish, before its - /// routes and before the later firing. - #[test] - fn a_positioned_record_follows_its_firings_finish_whatever_its_clock_says() { - let events = vec![ - finished(10, 1, 100), - routing(11, 1, 101), - applied(12, 1, 102), - started(13, 2, 103), - finished(14, 2, 104), - ]; - let records = vec![checkpoint(1, 1, 250)]; - let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "execution 0/10/0", - "platform 1", - "execution 0/11/0", - "execution 0/12/0", - "execution 0/13/0", - "execution 0/14/0", - ]); - } - - /// With the firing finished in an earlier pass, the record goes before - /// the first event of a later firing; a record with no position keeps - /// its clock order. - #[test] - fn a_positioned_record_precedes_later_firings_and_an_unpositioned_one_keeps_its_clock() { - let events = vec![started(13, 2, 103), finished(14, 2, 104)]; - let records = vec![checkpoint(1, 1, 250)]; - let finished_before: BTreeSet = [FiringKey::new(0, 1)].into_iter().collect(); - let (items, held) = order_items(&events, &records, &finished_before, false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "platform 1", - "execution 0/13/0", - "execution 0/14/0", - ]); - - let unpositioned = StoredPlatformRecord { - position: None, - ..checkpoint(2, 1, 250) - }; - let unpositioned = [unpositioned]; - let (items, held) = order_items(&events, &unpositioned, &BTreeSet::new(), false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "execution 0/13/0", - "execution 0/14/0", - "platform 2", - ]); - } - - /// A record whose firing has no finish yet, in the stream or in the pass, - /// is held back with everything after it until the finish arrives, or - /// until the run has finished. - #[test] - fn a_positioned_record_is_held_until_its_firings_finish_is_in_the_stream() { - let events = vec![started(13, 2, 103), finished(14, 2, 104)]; - let records = vec![ - checkpoint(1, 2, 50), - checkpoint(2, 3, 60), - checkpoint(3, 2, 70), - ]; - let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); - assert_eq!( - held, 2, - "the record for firing 3 holds itself and the one after it" - ); - assert_eq!(names(&items), vec![ - "execution 0/13/0", - "execution 0/14/0", - "platform 1", - ]); - - // Once the run finished, a record for a firing that never finished - // keeps its clock order; the firing's own records still follow its - // finish, in their seq order. - let (items, held) = order_items(&events, &records, &BTreeSet::new(), true); - assert_eq!(held, 0, "nothing is held once the run finished"); - assert_eq!(names(&items), vec![ - "platform 2", - "execution 0/13/0", - "execution 0/14/0", - "platform 1", - "platform 3", - ]); - } -} diff --git a/lib/components/fabro-petri/src/projector/order.rs b/lib/components/fabro-petri/src/projector/order.rs new file mode 100644 index 000000000..2b27aaff4 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/order.rs @@ -0,0 +1,343 @@ +//! The order one pass streams its new items in. + +use std::collections::BTreeSet; + +use fabro_store::platform_records::StoredPlatformRecord; +use petri_execution::events::{EventSource, RunEvent}; +use petri_runtime::engine::Event; + +use crate::projection::{FiringKey, Item}; + +/// The order one pass streams its new items in, and how many platform +/// records it holds back for a later pass. +/// +/// Every item is first ordered by `recorded_at` (stable: the coordinator +/// log before an execution log before a platform record on a tie, and each +/// log's own order kept). A platform record that carries a Petri position +/// (a checkpoint, keyed on `(execution, firing)`) is then placed by that +/// position, not by its clock, because the server stamps the record and the +/// worker stamps Petri's records and the two clocks can tie or invert: +/// +/// - before the firing's first `routing.resolved` event in the pass, which is +/// right after the firing's finish (its `step.finished` and the +/// `visit.completed` attached to it) and before the next firing's +/// `visit.started`, which is attached to that routing record; +/// - else after the last event of the firing in the pass; +/// - else, when the firing finished in an earlier pass, before the first event +/// of a later firing (a larger firing id) in the same execution, or where its +/// `recorded_at` put it; +/// - else the record is held back, with every platform record after it, and the +/// pass consumes platform records only up to it. The hook that writes a +/// checkpoint record runs after the driver appended the attempt's finish, but +/// the driver's store writer flushes that record on its own schedule, so the +/// platform record can be committed before its firing's `step.finished`; +/// holding it keeps the stream's order the same live and on a rebuild. +/// Nothing is held once the run has recorded its finish. +/// +/// The rule reads only the pass's own items and the firings already +/// finished, so a record is never streamed before its firing's finish and +/// never after the firing's routes. +pub(crate) fn order_items<'a>( + events: &'a [RunEvent], + platform_records: &'a [StoredPlatformRecord], + finished_before: &BTreeSet, + run_finished: bool, +) -> (Vec>, usize) { + let finished_in_pass = |at: FiringKey| { + events.iter().any(|event| { + FiringKey::of_event(event) == Some(at) + && matches!(event.engine(), Some(Event::StepFinished { .. })) + }) + }; + let finished = |at: FiringKey| finished_in_pass(at) || finished_before.contains(&at); + // Platform records are consumed in seq order: the first one whose firing + // has not finished holds itself and everything after it. + let consumed = if run_finished { + platform_records.len() + } else { + platform_records + .iter() + .position(|record| { + record + .position + .is_some_and(|position| !finished(FiringKey::from(position))) + }) + .unwrap_or(platform_records.len()) + }; + let held = platform_records.len() - consumed; + let platform_records = &platform_records[..consumed]; + + let mut items: Vec<(u64, u8, Item<'a>)> = + Vec::with_capacity(events.len() + platform_records.len()); + for event in events { + let rank = match event.id.source { + EventSource::Coordinator => 0, + EventSource::Execution { .. } => 1, + }; + items.push((event.recorded_at, rank, Item::Petri(event))); + } + for record in platform_records { + items.push((record.recorded_at, 2, Item::Platform(record))); + } + items.sort_by_key(|(recorded_at, rank, _)| (*recorded_at, *rank)); + + let item_firing = |item: &Item<'a>| match item { + Item::Petri(event) => FiringKey::of_event(event), + Item::Platform(_) => None, + }; + let is_routing = |item: &Item<'a>| { + matches!( + item, + Item::Petri(event) if matches!(event.engine(), Some(Event::RoutingResolved { .. })) + ) + }; + // The key of each item: its index in clock order, and whether it sits + // before (0), at (1) or after (2) that index. + let mut keys: Vec<(usize, u8)> = (0..items.len()).map(|index| (index, 1)).collect(); + for (index, (_, _, item)) in items.iter().enumerate() { + let Item::Platform(record) = item else { + continue; + }; + let Some(position) = record.position else { + continue; + }; + let at = FiringKey::from(position); + let first_routing = items + .iter() + .position(|(_, _, other)| item_firing(other) == Some(at) && is_routing(other)); + let last_of_firing = items + .iter() + .rposition(|(_, _, other)| item_firing(other) == Some(at)); + let first_later = items.iter().position(|(_, _, other)| { + item_firing(other) + .is_some_and(|key| key.execution == at.execution && key.firing > at.firing) + }); + keys[index] = if let Some(before) = first_routing { + (before, 0) + } else if let Some(after) = last_of_firing { + (after, 2) + } else if let Some(before) = first_later { + (before, 0) + } else { + (index, 1) + }; + } + let mut order: Vec = (0..items.len()).collect(); + order.sort_by_key(|index| keys[*index]); + let mut ordered: Vec>> = + items.into_iter().map(|(_, _, item)| Some(item)).collect(); + let items = order + .into_iter() + .map(|index| ordered[index].take().expect("each item is placed once")) + .collect(); + (items, held) +} + +#[cfg(test)] +mod tests { + use fabro_store::PlatformRecord; + use fabro_store::platform_records::{CheckpointRecord, StagePosition}; + use petri_execution::events::{Context, EventId, NodeRef, Record, RecordOrigin, Subject}; + use petri_execution::{ExecutionId, StoredEngineRecord}; + use petri_runtime::driver::BranchRole; + use petri_runtime::engine::{DecisionId, EventOrigin, RouteApplied}; + use petri_runtime::ir::{Attempt, FiringId, NodeId, Outcome, Status}; + + use super::*; + use crate::projector::stream; + + /// A firing's engine event at `seq`, recorded at `at`. + fn engine_event(seq: u64, firing: u64, at: u64, body: Event) -> RunEvent { + RunEvent { + id: EventId { + source: EventSource::Execution { + execution: ExecutionId::new(0), + }, + seq, + index: 0, + }, + origin: RecordOrigin::External, + context: Context { + invocation: None, + execution: Some(ExecutionId::new(0)), + parent: None, + }, + subject: Some(Subject { + node: NodeRef { + id: NodeId::new(1), + name: format!("n{firing}").into(), + kind: "attractor/command".into(), + meta: serde_json::Value::Null, + }, + firing: Some(FiringId::new(firing)), + visit: Some(1), + attempt: Some(Attempt::FIRST), + generation: None, + branch: BranchRole::None, + }), + observed_at: None, + recorded_at: at, + record: Some(Record::Engine(StoredEngineRecord { + seq, + origin: EventOrigin::External, + recorded_at: at, + body, + })), + derived: None, + } + } + + fn finished(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::StepFinished { + firing: FiringId::new(firing), + attempt: Attempt::FIRST, + outcome: Outcome::new(Status::Success, serde_json::Value::Null), + }) + } + + fn routing(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::RoutingResolved { + decision_id: DecisionId::route(FiringId::new(firing), Attempt::FIRST), + groups: Vec::new(), + }) + } + + fn applied(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::RouteApplied { + applied: RouteApplied::None { + firing: FiringId::new(firing), + group: 0, + }, + }) + } + + fn started(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::StepStarted { + firing: FiringId::new(firing), + attempt: Attempt::FIRST, + }) + } + + fn checkpoint(seq: u64, firing: u64, at: u64) -> StoredPlatformRecord { + StoredPlatformRecord { + seq, + recorded_at: at, + record: PlatformRecord::Checkpoint(CheckpointRecord { + execution: 0, + firing, + attempt: Some(1), + workspace: None, + git_commit_sha: Some("abc".to_string()), + diff_summary: None, + patch_blob: None, + operation: None, + }), + position: Some(StagePosition { + execution: 0, + firing, + }), + } + } + + fn names(items: &[Item<'_>]) -> Vec { + items + .iter() + .map(|item| match item { + Item::Petri(event) => stream::event_id_text(&event.id), + Item::Platform(record) => format!("platform {}", record.seq), + }) + .collect() + } + + /// A firing's events, then a later firing's events, then a checkpoint + /// for the first firing stamped later than all of them: the stream puts + /// the checkpoint right after the first firing's finish, before its + /// routes and before the later firing. + #[test] + fn a_positioned_record_follows_its_firings_finish_whatever_its_clock_says() { + let events = vec![ + finished(10, 1, 100), + routing(11, 1, 101), + applied(12, 1, 102), + started(13, 2, 103), + finished(14, 2, 104), + ]; + let records = vec![checkpoint(1, 1, 250)]; + let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "execution 0/10/0", + "platform 1", + "execution 0/11/0", + "execution 0/12/0", + "execution 0/13/0", + "execution 0/14/0", + ]); + } + + /// With the firing finished in an earlier pass, the record goes before + /// the first event of a later firing; a record with no position keeps + /// its clock order. + #[test] + fn a_positioned_record_precedes_later_firings_and_an_unpositioned_one_keeps_its_clock() { + let events = vec![started(13, 2, 103), finished(14, 2, 104)]; + let records = vec![checkpoint(1, 1, 250)]; + let finished_before: BTreeSet = [FiringKey::new(0, 1)].into_iter().collect(); + let (items, held) = order_items(&events, &records, &finished_before, false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "platform 1", + "execution 0/13/0", + "execution 0/14/0", + ]); + + let unpositioned = StoredPlatformRecord { + position: None, + ..checkpoint(2, 1, 250) + }; + let unpositioned = [unpositioned]; + let (items, held) = order_items(&events, &unpositioned, &BTreeSet::new(), false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "execution 0/13/0", + "execution 0/14/0", + "platform 2", + ]); + } + + /// A record whose firing has no finish yet, in the stream or in the pass, + /// is held back with everything after it until the finish arrives, or + /// until the run has finished. + #[test] + fn a_positioned_record_is_held_until_its_firings_finish_is_in_the_stream() { + let events = vec![started(13, 2, 103), finished(14, 2, 104)]; + let records = vec![ + checkpoint(1, 2, 50), + checkpoint(2, 3, 60), + checkpoint(3, 2, 70), + ]; + let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); + assert_eq!( + held, 2, + "the record for firing 3 holds itself and the one after it" + ); + assert_eq!(names(&items), vec![ + "execution 0/13/0", + "execution 0/14/0", + "platform 1", + ]); + + // Once the run finished, a record for a firing that never finished + // keeps its clock order; the firing's own records still follow its + // finish, in their seq order. + let (items, held) = order_items(&events, &records, &BTreeSet::new(), true); + assert_eq!(held, 0, "nothing is held once the run finished"); + assert_eq!(names(&items), vec![ + "platform 2", + "execution 0/13/0", + "execution 0/14/0", + "platform 1", + "platform 3", + ]); + } +} diff --git a/lib/components/fabro-petri/src/projector/signalling.rs b/lib/components/fabro-petri/src/projector/signalling.rs new file mode 100644 index 000000000..c81a97587 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/signalling.rs @@ -0,0 +1,99 @@ +//! The in-process wake-up: a run store whose appends signal the projector, +//! for a run that executes in the same process as the projector, over the +//! SQLite store directly, where no append endpoint is there to signal. + +use std::sync::Arc; + +use fabro_types::RunId; +use petri_execution::{Access, RunKey}; +use petri_store::StoreError; + +use super::Projector; +use crate::projection; + +impl Projector { + /// A run store whose appends signal this projector: for a run that + /// executes in the same process as the projector, over the SQLite store + /// directly, where no append endpoint is there to signal. The signal is + /// sent after the store's append returned, so the records it covers are + /// durable before the view sees them. + pub fn observe_store( + self: &Arc, + inner: Arc, + ) -> Arc { + Arc::new(SignallingStore { + inner, + projector: Arc::clone(self), + }) + } +} + +/// A run store that signals a projector after each append. +struct SignallingStore { + inner: Arc, + projector: Arc, +} + +#[async_trait::async_trait] +impl petri_execution::RunStore for SignallingStore { + async fn open( + &self, + key: &RunKey, + access: Access, + ) -> Result, StoreError> { + let logs = self.inner.open(key, access).await?; + Ok(Arc::new(SignallingLogs { + inner: logs, + run_id: projection::run_id_of(key.as_str()), + projector: Arc::clone(&self.projector), + })) + } +} + +struct SignallingLogs { + inner: Arc, + run_id: Option, + projector: Arc, +} + +#[async_trait::async_trait] +impl petri_execution::RunLogs for SignallingLogs { + fn locator(&self) -> String { + self.inner.locator() + } + + async fn append( + &self, + log: &petri_execution::LogId, + records: &[petri_execution::Record], + ) -> Result<(), StoreError> { + self.inner.append(log, records).await?; + if let Some(run_id) = self.run_id { + self.projector.signal(run_id); + } + Ok(()) + } + + async fn read( + &self, + log: &petri_execution::LogId, + ) -> Result, StoreError> { + self.inner.read(log).await + } + + async fn read_from( + &self, + log: &petri_execution::LogId, + seq: u64, + ) -> Result, StoreError> { + self.inner.read_from(log, seq).await + } + + async fn put_blob(&self, bytes: &[u8]) -> Result { + self.inner.put_blob(bytes).await + } + + async fn get_blob(&self, digest: petri_store::Digest) -> Result>, StoreError> { + self.inner.get_blob(digest).await + } +} diff --git a/lib/components/fabro-petri/src/projector/stream.rs b/lib/components/fabro-petri/src/projector/stream.rs new file mode 100644 index 000000000..f35e63d10 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/stream.rs @@ -0,0 +1,113 @@ +//! The run's stream as the view tables hold it: one row per folded item, +//! written by a pass and read back in `stream_seq` order. + +use fabro_db::DbPool; +use fabro_types::{RunId, RunStreamItem, RunStreamItemKind}; +use petri_execution::events::{EventId, EventSource}; + +use super::{Positions, ProjectError}; +use crate::projection::{Item, RunView}; + +pub(crate) struct StreamRow { + pub(super) stream_seq: u64, + pub(super) item_kind: &'static str, + pub(super) item_id: String, + pub(super) event_json: String, +} + +/// Fold the items into the view in order, each at the next delivery +/// sequence, advancing the positions with each, and produce its stream +/// row. +pub(crate) fn stream_rows( + items: &[Item<'_>], + view: &mut RunView, + positions: &mut Positions, + stream_seq: &mut u64, +) -> Result, ProjectError> { + let mut rows = Vec::with_capacity(items.len()); + for item in items { + *stream_seq += 1; + view.fold(item, *stream_seq); + let row = match item { + Item::Petri(event) => { + positions.advance(event.id); + StreamRow { + stream_seq: *stream_seq, + item_kind: "petri", + item_id: event_id_text(&event.id), + event_json: serde_json::to_string(event).map_err(ProjectError::Encode)?, + } + } + Item::Platform(record) => { + positions.platform_seq = record.seq; + StreamRow { + stream_seq: *stream_seq, + item_kind: "platform", + item_id: record.seq.to_string(), + event_json: serde_json::to_string(record).map_err(ProjectError::Encode)?, + } + } + }; + rows.push(row); + } + Ok(rows) +} + +/// The run's stream past the cursor, read from the view tables: up to +/// `limit` rows with `stream_seq > after`, in order, in Fabro's envelope. +pub(super) async fn stream_after( + views: &DbPool, + run_id: RunId, + after: u64, + limit: usize, +) -> Result, ProjectError> { + let rows: Vec<(i64, String, String, String)> = sqlx::query_as( + "SELECT stream_seq, item_kind, item_id, event_json FROM petri_stream WHERE run_id = ? AND \ + stream_seq > ? ORDER BY stream_seq LIMIT ?", + ) + .bind(run_id.to_string()) + .bind(column(after)) + .bind(i64::try_from(limit).unwrap_or(i64::MAX)) + .fetch_all(views) + .await + .map_err(ProjectError::Database)?; + rows.into_iter() + .map(|(stream_seq, item_kind, item_id, event_json)| { + let item: serde_json::Value = + serde_json::from_str(&event_json).map_err(ProjectError::Encode)?; + let kind = match item_kind.as_str() { + "platform" => RunStreamItemKind::Platform, + _ => RunStreamItemKind::Petri, + }; + let recorded_at = item + .get("recorded_at") + .and_then(serde_json::Value::as_u64) + .unwrap_or(0); + Ok(RunStreamItem { + run_id, + stream_seq: u64::try_from(stream_seq).unwrap_or(0), + kind, + id: item_id, + recorded_at, + item, + }) + }) + .collect() +} + +/// A Petri event id as the stream names it: `//`. +#[must_use] +pub(crate) fn event_id_text(id: &EventId) -> String { + format!("{}/{}/{}", log_text(&id.source), id.seq, id.index) +} + +pub(super) fn log_text(source: &EventSource) -> String { + match source { + EventSource::Coordinator => "coordinator".to_string(), + EventSource::Execution { execution } => format!("execution {execution}"), + } +} + +pub(super) fn column(value: u64) -> i64 { + i64::try_from(value).unwrap_or(i64::MAX) +} diff --git a/lib/components/fabro-petri/src/test_support.rs b/lib/components/fabro-petri/src/test_support.rs index 38b37ce07..91793f60e 100644 --- a/lib/components/fabro-petri/src/test_support.rs +++ b/lib/components/fabro-petri/src/test_support.rs @@ -1,24 +1,35 @@ //! Petri's test kit, for Fabro crates that check a store implementation -//! against Petri's contract from their own tests, an in-memory platform +//! against Petri's contract from their own tests; an in-memory platform //! record store and an in-memory blob table for tests of the hooks and -//! recovery. Compiled only with the `test-support` feature, which a +//! recovery; and the readers over the view tables a test compares a live +//! view with. Compiled only with the `test-support` feature, which a //! dev-dependency turns on. -use std::collections::HashMap; +use std::collections::{BTreeSet, HashMap}; use std::sync::Mutex; use std::time::Duration; use async_trait::async_trait; use bytes::Bytes; -use fabro_store::platform_records::now_ms; -use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord}; +use fabro_db::DbPool; +use fabro_store::platform_records::{PlatformRecordStore, now_ms}; +use fabro_store::{ + PlatformRecord, PlatformRecordKind, RunProjection, StagePosition, StoredPlatformRecord, +}; use fabro_types::{BlobHash, RunId}; +use fabro_util::error::collect_chain; use fabro_util::sync; +use petri_execution::events::{self, RunEvent}; +use petri_execution::{Access, CoordinatorEvent, RunKey, RunStore as _}; +use petri_store::StoreError; pub use petri_testkit::run_store; +use tracing::warn; +use crate::SqliteRunStore; use crate::blobs::Blobs; use crate::platform_records::{PlatformRecordError, PlatformRecords}; -use crate::projector::Projector; +use crate::projection::RunView; +use crate::projector::{self, Positions, ProjectError, Projector, order, stream}; /// Whether the projector keeps a cache for the run: the replay and the /// view its passes continue from. @@ -126,3 +137,100 @@ impl PlatformRecords for MemoryPlatformRecords { .collect()) } } + +/// The run's projection rebuilt from its records alone, with nothing +/// stored: what a fresh projector would commit over the same records. A test +/// compares it with the live view. `records` and `views` are the two pools +/// [`Projector::new`] takes. +pub async fn rebuild( + records: &DbPool, + views: &DbPool, + run_id: RunId, +) -> Result<(Option, Positions, u64), ProjectError> { + let store = SqliteRunStore::new(records.clone()); + let platform = PlatformRecordStore::new(views.clone()); + let key = RunKey::new(run_id.to_string()); + let platform_records = platform.read(&run_id).await.map_err(ProjectError::Store)?; + let events = match store.open(&key, Access::Read).await { + Ok(logs) => events::replay_run(&*logs) + .await + .inspect_err(|error| { + warn!(error = %collect_chain(error).join(": "), "rebuild: the run does not replay"); + }) + .unwrap_or_default(), + Err(StoreError::NotFound { .. }) => Vec::new(), + Err(error) => return Err(ProjectError::Open(error)), + }; + let run_finished = events.iter().any(|event| { + matches!( + event.coordinator(), + Some(CoordinatorEvent::RunFinished { .. }) + ) + }); + let (items, _held) = + order::order_items(&events, &platform_records, &BTreeSet::new(), run_finished); + let mut view = RunView::new(); + let mut positions = Positions::default(); + let mut stream_seq = 0; + stream::stream_rows(&items, &mut view, &mut positions, &mut stream_seq)?; + Ok((view.projection, positions, stream_seq)) +} + +/// The stored view's positions and stream sequence; `views` is the pool +/// the view tables live in. +pub async fn stored_positions( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + projector::stored_positions(views, run_id).await +} + +/// The stored view's projection, for a test or a reader outside the store. +pub async fn stored_projection( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + let json: Option = + sqlx::query_scalar("SELECT projection_json FROM petri_projection WHERE run_id = ?") + .bind(run_id.to_string()) + .fetch_optional(views) + .await + .map_err(ProjectError::Database)?; + json.map(|json| serde_json::from_str(&json).map_err(ProjectError::Encode)) + .transpose() +} + +/// The stream rows of a run: `(stream_seq, item_kind, item_id)`, in order. +pub async fn stored_stream( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + let rows: Vec<(i64, String, String)> = sqlx::query_as( + "SELECT stream_seq, item_kind, item_id FROM petri_stream WHERE run_id = ? ORDER BY stream_seq", + ) + .bind(run_id.to_string()) + .fetch_all(views) + .await + .map_err(ProjectError::Database)?; + Ok(rows + .into_iter() + .map(|(seq, kind, id)| (u64::try_from(seq).unwrap_or(0), kind, id)) + .collect()) +} + +/// Every stored platform record of a run, for a reader outside the store. +pub async fn stored_platform_records( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + PlatformRecordStore::new(views.clone()) + .read(&run_id) + .await + .map_err(ProjectError::Store) +} + +/// A recorded event's projection is what `RunEvent` serializes to. +#[must_use] +pub fn event_json(event: &RunEvent) -> serde_json::Value { + serde_json::to_value(event).unwrap_or_default() +} diff --git a/lib/components/fabro-petri/tests/projection.rs b/lib/components/fabro-petri/tests/projection.rs index 31f6e6ca1..aa7ff9914 100644 --- a/lib/components/fabro-petri/tests/projection.rs +++ b/lib/components/fabro-petri/tests/projection.rs @@ -418,15 +418,15 @@ fn json(value: &T) -> serde_json::Value { /// The stored view equals the view rebuilt from the records alone: the /// projection, the positions and the delivery sequence. async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { - let stored = projector::stored_projection(pool, run_id) + let stored = petri_support::stored_projection(pool, run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection"); - let (stored_positions, stored_stream_seq) = projector::stored_positions(pool, run_id) + let (stored_positions, stored_stream_seq) = petri_support::stored_positions(pool, run_id) .await .expect("the positions read") .expect("the run has positions"); - let (rebuilt, positions, stream_seq) = projector::rebuild(pool, pool, run_id) + let (rebuilt, positions, stream_seq) = petri_support::rebuild(pool, pool, run_id) .await .expect("the run rebuilds"); let rebuilt = rebuilt.expect("the rebuild has a projection"); @@ -442,7 +442,7 @@ async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { positions.petri.sort(); assert_eq!(stored_positions, positions); assert_eq!(stored_stream_seq, stream_seq); - let stream = projector::stored_stream(pool, run_id) + let stream = petri_support::stored_stream(pool, run_id) .await .expect("the stream reads"); let seqs: Vec = stream.iter().map(|(seq, _, _)| *seq).collect(); @@ -454,7 +454,7 @@ async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { } async fn stage_states(pool: &DbPool, run_id: RunId) -> Vec<(String, StageState)> { - let stored = projector::stored_projection(pool, run_id) + let stored = petri_support::stored_projection(pool, run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection"); @@ -472,7 +472,7 @@ async fn the_hello_bundle_projects_live_as_it_rebuilds() { let scenario = hello_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -501,7 +501,7 @@ async fn a_large_output_projects_as_its_blob_reference() { let scenario = large_output_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -525,7 +525,7 @@ async fn a_command_workflow_projects_live_as_it_rebuilds() { let scenario = command_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -554,7 +554,7 @@ async fn a_parallel_workflow_projects_its_branches_as_child_executions() { let scenario = parallel_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -593,7 +593,7 @@ async fn dropped_wake_ups_are_caught_up_by_the_next_signal() { let scenario = command_scenario().await; run_unobserved(&scenario).await; assert!( - projector::stored_projection(&scenario.pool, scenario.run_id) + petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .is_none(), @@ -780,12 +780,13 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( .expect("the first pass commits"); assert!(!pass.skipped); assert!(!pass.health.complete, "the run has not finished"); - let (positions_before, stream_before) = projector::stored_positions(&replayed, scenario.run_id) - .await - .expect("reads") - .expect("positions"); + let (positions_before, stream_before) = + petri_support::stored_positions(&replayed, scenario.run_id) + .await + .expect("reads") + .expect("positions"); assert_eq!(pass.stream_seq, stream_before); - let stream_rows_before = projector::stored_stream(&replayed, scenario.run_id) + let stream_rows_before = petri_support::stored_stream(&replayed, scenario.run_id) .await .expect("reads") .len(); @@ -801,7 +802,7 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( "{crashed:?}" ); assert_eq!( - projector::stored_positions(&replayed, scenario.run_id) + petri_support::stored_positions(&replayed, scenario.run_id) .await .expect("reads") .expect("positions"), @@ -813,11 +814,12 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( let after = Projector::new(replayed.clone(), replayed.clone()); let report = after.startup_pass().await.expect("the restart catches up"); assert_eq!((report.runs, report.projected), (1, 1)); - let (positions_after, stream_after) = projector::stored_positions(&replayed, scenario.run_id) - .await - .expect("reads") - .expect("positions"); - let stream_rows_after = projector::stored_stream(&replayed, scenario.run_id) + let (positions_after, stream_after) = + petri_support::stored_positions(&replayed, scenario.run_id) + .await + .expect("reads") + .expect("positions"); + let stream_rows_after = petri_support::stored_stream(&replayed, scenario.run_id) .await .expect("reads"); // Only the suffix was applied: the stream grew by the suffix's events, @@ -845,11 +847,11 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( // And the copy agrees with the run projected in one go over the source. let source = Projector::new(scenario.pool.clone(), scenario.pool.clone()); source.startup_pass().await.expect("the source projects"); - let whole = projector::stored_projection(&scenario.pool, scenario.run_id) + let whole = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); - let pieced = projector::stored_projection(&replayed, scenario.run_id) + let pieced = petri_support::stored_projection(&replayed, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -904,11 +906,11 @@ async fn a_restarted_projector_agrees_over_nested_child_executions() { let whole = Projector::new(scenario.pool.clone(), scenario.pool.clone()); whole.startup_pass().await.expect("the source projects"); - let one_go = projector::stored_projection(&scenario.pool, scenario.run_id) + let one_go = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); - let restarted = projector::stored_projection(&staged, scenario.run_id) + let restarted = petri_support::stored_projection(&staged, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -937,11 +939,11 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { .await .expect("the clean pass commits"); assert!(clean.health.complete, "{:?}", clean.health.incomplete); - let (positions, stream_seq) = projector::stored_positions(&scenario.pool, scenario.run_id) + let (positions, stream_seq) = petri_support::stored_positions(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("positions"); - let before = projector::stored_projection(&scenario.pool, scenario.run_id) + let before = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -973,7 +975,7 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { held.health ); let (positions_after, stream_after) = - projector::stored_positions(&scenario.pool, scenario.run_id) + petri_support::stored_positions(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("positions"); @@ -982,7 +984,7 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { "the view did not advance past the tear" ); assert_eq!(stream_after, stream_seq); - let after = projector::stored_projection(&scenario.pool, scenario.run_id) + let after = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -1240,9 +1242,10 @@ impl GateRun { async fn pending(&self) -> fabro_types::RunProjection { let deadline = Instant::now() + Duration::from_secs(30); loop { - let stored = projector::stored_projection(&self.scenario.pool, self.scenario.run_id) - .await - .expect("the stored projection reads"); + let stored = + petri_support::stored_projection(&self.scenario.pool, self.scenario.run_id) + .await + .expect("the stored projection reads"); if let Some(stored) = stored.filter(|stored| !stored.pending_interviews.is_empty()) { return stored; } @@ -1255,7 +1258,7 @@ impl GateRun { } async fn stored(&self) -> fabro_types::RunProjection { - projector::stored_projection(&self.scenario.pool, self.scenario.run_id) + petri_support::stored_projection(&self.scenario.pool, self.scenario.run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection") @@ -1282,7 +1285,7 @@ async fn an_expired_question_is_pending_while_the_gate_waits_and_closes_on_the_e let run_id = gate.scenario.run_id; let deadline = Instant::now() + Duration::from_secs(30); loop { - let stored = projector::stored_projection(&pool, run_id) + let stored = petri_support::stored_projection(&pool, run_id) .await .expect("the stored projection reads"); if let Some(stored) = stored.filter(|stored| !stored.pending_interviews.is_empty()) { From 12483e081e5f45bc022709c1673a0a0e6b309233 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:15:11 -0400 Subject: [PATCH 10/16] Plan recovery through a Planner and share the run's Git settings with the hooks RecoveryRequest::for_run and HooksSpec::for_run derived the author, the identity source, the checkpoint settings and host_workspaces from the run namespace with the same expressions. RunGitSettings, in checkpoint.rs beside RunWorkspaces, is that derivation once; both specs carry it. recovery::plan nested the "recorded, else found and reconciled" lookup two loops deep and tracked a found flag over a tuple list. A Planner holds the records, the workspace lookup and the recorded checkpoints, and its methods read in order: targets, snapshot_of, reconcile_record, newest. Candidate names the (key, sha) pair, and the Target struct that duplicated its key's execution is gone; last_finish yields the CheckpointKey itself, whose Display the failure reason now uses. Co-Authored-By: Claude Fable 5.1 --- lib/components/fabro-petri/src/checkpoint.rs | 52 ++- lib/components/fabro-petri/src/hooks.rs | 66 ++-- lib/components/fabro-petri/src/recovery.rs | 379 ++++++++++--------- lib/components/fabro-petri/tests/hooks.rs | 30 +- 4 files changed, 288 insertions(+), 239 deletions(-) diff --git a/lib/components/fabro-petri/src/checkpoint.rs b/lib/components/fabro-petri/src/checkpoint.rs index df82a04e6..85a186f0d 100644 --- a/lib/components/fabro-petri/src/checkpoint.rs +++ b/lib/components/fabro-petri/src/checkpoint.rs @@ -43,8 +43,8 @@ use std::time::Duration; use fabro_checkpoint::author::GitAuthor; use fabro_checkpoint::trailer::{self, Trailer}; use fabro_store::platform_records::{DecisionRef, OperationKey}; -use fabro_types::DiffSummary; -use fabro_types::settings::run::RunCheckpointSettings; +use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace}; +use fabro_types::{DiffSummary, GitIdentitySource, SandboxProviderKind}; use petri_runtime::executor::{EnvError, ExecEnv, OutputMode, ProcessSpec, Sig}; use petri_runtime::ir::LogStream; use tokio::process::Command; @@ -92,6 +92,54 @@ pub const EXCLUDE_DIRS: &[&str] = &[ ".pytest_cache", ]; +/// The settings a run's Git work runs under, as its namespace gives them: +/// who authors the checkpoint commits and where that identity came from, +/// the checkpoint settings, and whether the sandbox provider keeps the +/// workspaces on this host. The hooks and recovery both start from it. +#[derive(Clone, Debug)] +pub struct RunGitSettings { + pub author: GitAuthor, + pub identity_source: GitIdentitySource, + pub checkpoint: RunCheckpointSettings, + /// Whether the run's workspaces are on this host (the local sandbox + /// provider). A run elsewhere snapshots inside its sandboxes. + pub host_workspaces: bool, +} + +impl From<&RunNamespace> for RunGitSettings { + fn from(settings: &RunNamespace) -> Self { + let author = settings + .git + .author + .as_ref() + .map(GitAuthor::from) + .unwrap_or_default(); + let identity_source = if author.is_default() { + GitIdentitySource::Default + } else { + GitIdentitySource::Explicit + }; + Self { + author, + identity_source, + checkpoint: settings.checkpoint.clone(), + host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, + } + } +} + +impl Default for RunGitSettings { + /// Fabro's default author and checkpoint settings, on this host. + fn default() -> Self { + Self { + author: GitAuthor::default(), + identity_source: GitIdentitySource::Default, + checkpoint: RunCheckpointSettings::default(), + host_workspaces: true, + } + } +} + /// The identity of one snapshot: the attempt whose files it holds. #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] pub struct CheckpointKey { diff --git a/lib/components/fabro-petri/src/hooks.rs b/lib/components/fabro-petri/src/hooks.rs index 36914d942..7fa93e192 100644 --- a/lib/components/fabro-petri/src/hooks.rs +++ b/lib/components/fabro-petri/src/hooks.rs @@ -76,15 +76,12 @@ use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex, OnceLock}; use std::time::Duration; -use fabro_checkpoint::author::GitAuthor; use fabro_store::platform_records::{ ArtifactCollectedRecord, CheckpointRecord, GitIdentityRecord, RunBranchRecord, RunDiffRecord, }; use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition}; -use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace}; -use fabro_types::{ - BlobHash, DiffSummary, GitIdentity, GitIdentitySource, RunId, SandboxProviderKind, -}; +use fabro_types::settings::run::RunNamespace; +use fabro_types::{BlobHash, DiffSummary, GitIdentity, RunId}; use fabro_util::error::collect_chain; use fabro_util::sync; use fabro_util::workspace_glob::{WorkspaceGlobError, WorkspaceGlobSet}; @@ -103,8 +100,8 @@ use tracing::{debug, info, warn}; use crate::blobs::Blobs; use crate::checkpoint::{ - CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, EXCLUDE_DIRS, RunWorkspaces, Site, - Snapshot, WorkspaceDiff, + CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, EXCLUDE_DIRS, RunGitSettings, + RunWorkspaces, Site, Snapshot, WorkspaceDiff, }; use crate::platform_records::{PlatformRecordError, PlatformRecords}; use crate::recovery::{self, Plan, RecoveryError, RestoreTarget}; @@ -213,50 +210,27 @@ impl HookError { } /// What Fabro's hooks need beside the run: where the platform records go, -/// who authors the commits, the checkpoint settings, and which files are -/// the run's artifacts. +/// the run's Git settings, and which files are the run's artifacts. pub struct HooksSpec { - pub records: Arc, - pub author: GitAuthor, - /// Where the author identity came from: the run's settings, or Fabro's - /// default. - pub identity_source: GitIdentitySource, - pub checkpoint: RunCheckpointSettings, + pub records: Arc, + pub git: RunGitSettings, /// The `[run.artifacts] include` patterns: which files of a stage's /// workspace are collected after the stage. - pub artifacts: Vec, - /// Whether the run's workspaces are on this host (the local sandbox - /// provider). A run elsewhere snapshots inside its sandboxes. - pub host_workspaces: bool, + pub artifacts: Vec, /// A test's gate directory: a checkpoint point named by a `.hold` file /// there waits for its `.release` file. `None` outside tests. - pub test_gates: Option, + pub test_gates: Option, } impl HooksSpec { - /// The spec a run's settings give: its Git author, its checkpoint - /// settings, its artifact patterns, and whether its sandbox provider - /// keeps workspaces on this host. + /// The spec a run's settings give: its Git settings and its artifact + /// patterns. #[must_use] pub fn for_run(records: Arc, settings: &RunNamespace) -> Self { - let author = settings - .git - .author - .as_ref() - .map(GitAuthor::from) - .unwrap_or_default(); - let identity_source = if author.is_default() { - GitIdentitySource::Default - } else { - GitIdentitySource::Explicit - }; Self { records, - author, - identity_source, - checkpoint: settings.checkpoint.clone(), + git: RunGitSettings::from(settings), artifacts: settings.artifacts.include.clone(), - host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, test_gates: None, } } @@ -432,12 +406,16 @@ impl FabroHooks { blobs: Option>, ) -> Self { let identity = GitIdentity { - name: spec.author.name.clone(), - email: spec.author.email.clone(), - source: spec.identity_source, + name: spec.git.author.name.clone(), + email: spec.git.author.email.clone(), + source: spec.git.identity_source, }; - let workspaces = - RunWorkspaces::new(run_dir, run_id.to_string(), spec.author, &spec.checkpoint); + let workspaces = RunWorkspaces::new( + run_dir, + run_id.to_string(), + spec.git.author, + &spec.git.checkpoint, + ); Self { inner, run_id, @@ -446,7 +424,7 @@ impl FabroHooks { workspaces, lookup: WorkspaceLookup::new(Arc::clone(&store), run_key), identity, - host_workspaces: spec.host_workspaces, + host_workspaces: spec.git.host_workspaces, test_gates: spec.test_gates, handle: OnceLock::new(), checkpoints: CheckpointLedger::default(), diff --git a/lib/components/fabro-petri/src/recovery.rs b/lib/components/fabro-petri/src/recovery.rs index 76d56d7ea..ee62ec941 100644 --- a/lib/components/fabro-petri/src/recovery.rs +++ b/lib/components/fabro-petri/src/recovery.rs @@ -37,11 +37,10 @@ use std::collections::BTreeMap; use std::path::PathBuf; use std::sync::Arc; -use fabro_checkpoint::author::GitAuthor; use fabro_store::platform_records::CheckpointRecord; use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition}; -use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace}; -use fabro_types::{RunId, SandboxProviderKind}; +use fabro_types::RunId; +use fabro_types::settings::run::RunNamespace; use petri_execution::host::{self, HostError}; use petri_execution::inspect::{self, ExecutionInspection, InspectError}; use petri_execution::{Access, InvocationId, RunKey, RunStore}; @@ -49,28 +48,24 @@ use petri_store::StoreError; use tracing::info; use crate::checkpoint::{ - CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces, Site, + CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunGitSettings, RunWorkspaces, Site, }; use crate::platform_records::{PlatformRecordError, PlatformRecords}; use crate::workspace::{WorkspaceLookup, WorkspaceLookupError}; -/// What recovery needs: the run, where its workspaces are, its records. +/// What recovery needs: the run, where its workspaces are, its records, +/// and its Git settings. pub struct RecoveryRequest { - pub run_id: RunId, + pub run_id: RunId, /// The run directory Petri ran under (the run's `petri` scratch). - pub run_dir: PathBuf, - pub store: Arc, - pub records: Arc, - pub author: GitAuthor, - pub checkpoint: RunCheckpointSettings, - /// Whether the run's workspaces are on this host. - pub host_workspaces: bool, + pub run_dir: PathBuf, + pub store: Arc, + pub records: Arc, + pub git: RunGitSettings, } impl RecoveryRequest { - /// The request a run's settings give: its Git author, its checkpoint - /// settings, and whether its sandbox provider keeps workspaces on this - /// host. + /// The request a run's settings give. #[must_use] pub fn for_run( run_id: RunId, @@ -84,14 +79,7 @@ impl RecoveryRequest { run_dir, store, records, - author: settings - .git - .author - .as_ref() - .map(GitAuthor::from) - .unwrap_or_default(), - checkpoint: settings.checkpoint.clone(), - host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, + git: RunGitSettings::from(settings), } } } @@ -173,12 +161,6 @@ pub enum RecoveryError { }, } -/// One live execution's last durable finish and the snapshot it names. -struct Target { - execution: u64, - key: CheckpointKey, -} - /// Decide how the run continues: the snapshot every live workspace must sit /// on, from the records and the snapshot repository, with a lost record /// reconciled from the repository. Nothing is touched. @@ -221,80 +203,14 @@ pub async fn plan( if let Some(failed) = checkpoint_failure(&inspection.executions) { return Ok(Plan::Failed { reason: failed }); } - - let lookup = WorkspaceLookup::new(Arc::clone(&store), key); - let recorded = recorded_checkpoints(records, run_id).await?; - - // The snapshot each live execution's workspace must sit on. A live - // execution is one whose log records no exit: `inspect_run` reports it - // as incomplete. - let mut candidates: BTreeMap> = BTreeMap::new(); - for execution in inspection - .executions - .iter() - .filter(|execution| execution.status == "incomplete") - { - let Some(target) = last_finish(execution) else { - continue; - }; - let owned = lookup - .of_invocation(execution.invocation) - .await - .map_err(RecoveryError::Lookup)?; - if owned.is_empty() { - continue; - } - let mut found = false; - for workspace in owned { - let sha = match recorded.get(&target.key) { - Some((recorded_workspace, sha)) - if recorded_workspace.as_deref().is_none_or(|w| w == workspace) => - { - Some(sha.clone()) - } - _ => { - let sha = workspaces - .find(&workspace, target.key) - .await - .map_err(|source| RecoveryError::Workspace { - workspace: workspace.clone(), - source, - })?; - if let Some(sha) = &sha { - reconcile_record(records, run_id, target.key, &workspace, sha).await?; - } - sha - } - }; - if let Some(sha) = sha { - found = true; - candidates - .entry(workspace) - .or_default() - .push((Target { ..target }, sha)); - } - } - if !found { - return Ok(Plan::Failed { - reason: format!( - "no checkpoint snapshot exists for the last durable finish of execution {} \ - (firing {} attempt {}); the run cannot resume on stale files", - target.execution, target.key.firing, target.key.attempt - ), - }); - } - } - - let mut targets = BTreeMap::new(); - for (workspace, candidates) in candidates { - let sha = newest(workspaces, &workspace, &candidates).await?; - let key = candidates - .iter() - .find(|(_, candidate)| *candidate == sha) - .map_or(candidates[0].0.key, |(target, _)| target.key); - targets.insert(workspace, RestoreTarget { key, sha }); - } - Ok(Plan::Resume { targets }) + let planner = Planner { + records, + run_id, + workspaces, + lookup: WorkspaceLookup::new(store, key), + recorded: recorded_checkpoints(records, run_id).await?, + }; + planner.targets(&inspection.executions).await } /// Decide how the run continues, and bring its host workspaces to their @@ -303,8 +219,8 @@ pub async fn recover(request: RecoveryRequest) -> Result Result Result { + records: &'a dyn PlatformRecords, + run_id: &'a RunId, + workspaces: &'a RunWorkspaces, + lookup: WorkspaceLookup, + /// The run's checkpoint records by key: the workspace they name and + /// the commit. + recorded: BTreeMap, String)>, +} + +impl Planner<'_> { + /// The snapshot each live execution's workspace must sit on. A live + /// execution is one whose log records no exit: `inspect_run` reports + /// it as incomplete. A workspace several live executions share is + /// brought to the newest of their snapshots. + async fn targets(&self, executions: &[ExecutionInspection]) -> Result { + let mut candidates: BTreeMap> = BTreeMap::new(); + for execution in executions + .iter() + .filter(|execution| execution.status == "incomplete") + { + let Some(key) = last_finish(execution) else { + continue; + }; + let owned = self + .lookup + .of_invocation(execution.invocation) + .await + .map_err(RecoveryError::Lookup)?; + if owned.is_empty() { + continue; + } + let mut found = false; + for workspace in owned { + if let Some(sha) = self.snapshot_of(&workspace, key).await? { + found = true; + candidates + .entry(workspace) + .or_default() + .push(Candidate { key, sha }); + } + } + if !found { + return Ok(Plan::Failed { + reason: format!( + "no checkpoint snapshot exists for the last durable finish of {key}; the \ + run cannot resume on stale files" + ), + }); + } + } + + let mut targets = BTreeMap::new(); + for (workspace, candidates) in candidates { + let target = self.newest(&workspace, &candidates).await?; + targets.insert(workspace, target); + } + Ok(Plan::Resume { targets }) + } + + /// The snapshot of `key` in `workspace`: the commit its record names, + /// when the record names this workspace or none; else the commit found + /// by its key in the workspace's snapshot repository or history, which + /// is then recorded again for the record the crash lost. `None` when no + /// snapshot exists. + async fn snapshot_of( + &self, + workspace: &str, + key: CheckpointKey, + ) -> Result, RecoveryError> { + if let Some((recorded_workspace, sha)) = self.recorded.get(&key) { + if recorded_workspace + .as_deref() + .is_none_or(|recorded| recorded == workspace) + { + return Ok(Some(sha.clone())); + } + } + let found = self + .workspaces + .find(workspace, key) + .await + .map_err(|source| RecoveryError::Workspace { + workspace: workspace.to_string(), + source, + })?; + if let Some(sha) = &found { + self.reconcile_record(key, workspace, sha).await?; + } + Ok(found) + } + + /// Write the record a crash lost, from the commit found by its key. + async fn reconcile_record( + &self, + key: CheckpointKey, + workspace: &str, + sha: &str, + ) -> Result<(), RecoveryError> { + info!( + run_id = %self.run_id, + execution = key.execution, + firing = key.firing, + attempt = key.attempt, + sha, + "checkpoint record reconciled from the run branch" + ); + let record = PlatformRecord::Checkpoint(CheckpointRecord { + execution: key.execution, + firing: key.firing, + attempt: Some(key.attempt), + workspace: Some(workspace.to_string()), + git_commit_sha: Some(sha.to_string()), + diff_summary: None, + patch_blob: None, + operation: Some(key.operation()), + }); + self.records + .append( + self.run_id, + &record, + Some(StagePosition { + execution: key.execution, + firing: key.firing, + }), + ) + .await + .map_err(RecoveryError::Records)?; + Ok(()) + } + + /// Of the snapshots live executions name on one workspace, the one + /// every other descends from, else the last named. Two executions that + /// name the same commit share it under the first one's key. + async fn newest( + &self, + workspace: &str, + candidates: &[Candidate], + ) -> Result { + let mut chosen = &candidates[0]; + for candidate in &candidates[1..] { + if self + .workspaces + .is_ancestor(workspace, &chosen.sha, &candidate.sha) + .await + .map_err(|source| RecoveryError::Workspace { + workspace: workspace.to_string(), + source, + })? + { + chosen = candidate; + } + } + let chosen = candidates + .iter() + .find(|candidate| candidate.sha == chosen.sha) + .unwrap_or(chosen); + Ok(RestoreTarget { + key: chosen.key, + sha: chosen.sha.clone(), + }) + } +} + /// The reason a run with a failed checkpoint is reported failed, when it /// has one. fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option { @@ -375,16 +463,14 @@ fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option { }) } -/// The last `StepFinished` of an execution's log. -fn last_finish(execution: &ExecutionInspection) -> Option { +/// The last `StepFinished` of an execution's log, as the key of its +/// snapshot. +fn last_finish(execution: &ExecutionInspection) -> Option { let attempt = execution.engine.as_ref()?.attempts.last()?; - Some(Target { + Some(CheckpointKey { execution: execution.execution.raw(), - key: CheckpointKey { - execution: execution.execution.raw(), - firing: attempt.firing, - attempt: attempt.attempt, - }, + firing: attempt.firing, + attempt: attempt.attempt, }) } @@ -414,69 +500,6 @@ async fn recorded_checkpoints( Ok(recorded) } -/// Write the record a crash lost, from the commit found by its key. -async fn reconcile_record( - records: &dyn PlatformRecords, - run_id: &RunId, - key: CheckpointKey, - workspace: &str, - sha: &str, -) -> Result<(), RecoveryError> { - info!( - run_id = %run_id, - execution = key.execution, - firing = key.firing, - attempt = key.attempt, - sha, - "checkpoint record reconciled from the run branch" - ); - let record = PlatformRecord::Checkpoint(CheckpointRecord { - execution: key.execution, - firing: key.firing, - attempt: Some(key.attempt), - workspace: Some(workspace.to_string()), - git_commit_sha: Some(sha.to_string()), - diff_summary: None, - patch_blob: None, - operation: Some(key.operation()), - }); - records - .append( - run_id, - &record, - Some(StagePosition { - execution: key.execution, - firing: key.firing, - }), - ) - .await - .map_err(RecoveryError::Records)?; - Ok(()) -} - -/// Of the snapshots live executions name on one workspace, the one every -/// other descends from, else the last named. -async fn newest( - workspaces: &RunWorkspaces, - workspace: &str, - targets: &[(Target, String)], -) -> Result { - let mut chosen = &targets[0].1; - for (_, sha) in &targets[1..] { - if workspaces - .is_ancestor(workspace, chosen, sha) - .await - .map_err(|source| RecoveryError::Workspace { - workspace: workspace.to_string(), - source, - })? - { - chosen = sha; - } - } - Ok(chosen.clone()) -} - /// Verify, reset or restore the workspace at `site` onto its target: a /// workspace that still holds the commit is verified or reset in place; a /// gone one, a fresh directory with no history (a fork's first diff --git a/lib/components/fabro-petri/tests/hooks.rs b/lib/components/fabro-petri/tests/hooks.rs index 1ba189a95..5849b4633 100644 --- a/lib/components/fabro-petri/tests/hooks.rs +++ b/lib/components/fabro-petri/tests/hooks.rs @@ -24,7 +24,9 @@ use fabro_checkpoint::author::GitAuthor; use fabro_petri::admission::AdmittedGraphs; use fabro_petri::blobs::Blobs; use fabro_petri::check::{self, Bundle, CheckRequest, Launch}; -use fabro_petri::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces}; +use fabro_petri::checkpoint::{ + CHECKPOINT_FAILED_CLASS, CheckpointKey, RunGitSettings, RunWorkspaces, +}; use fabro_petri::controls::RunControls; use fabro_petri::engine::{self, Execution, RunRequest, RunStatus}; use fabro_petri::hooks::HooksSpec; @@ -166,13 +168,13 @@ impl Harness { fn hooks(&self, provider: &SandboxProviderKind) -> HooksSpec { HooksSpec { - records: Arc::clone(&self.records) as Arc, - author: GitAuthor::default(), - identity_source: GitIdentitySource::Default, - checkpoint: RunCheckpointSettings::default(), - artifacts: self.artifacts.clone(), - host_workspaces: *provider == SandboxProviderKind::LOCAL, - test_gates: None, + records: Arc::clone(&self.records) as Arc, + git: RunGitSettings { + host_workspaces: *provider == SandboxProviderKind::LOCAL, + ..RunGitSettings::default() + }, + artifacts: self.artifacts.clone(), + test_gates: None, } } @@ -285,13 +287,11 @@ impl Harness { async fn recover(&self) -> Recovery { recovery::recover(RecoveryRequest { - run_id: self.run_id, - run_dir: self.run_dir.clone(), - store: Arc::clone(&self.store) as Arc, - records: Arc::clone(&self.records) as Arc, - author: GitAuthor::default(), - checkpoint: RunCheckpointSettings::default(), - host_workspaces: true, + run_id: self.run_id, + run_dir: self.run_dir.clone(), + store: Arc::clone(&self.store) as Arc, + records: Arc::clone(&self.records) as Arc, + git: RunGitSettings::default(), }) .await .expect("recovery decides") From f471676802707a7a692d2e5e2ac2d92038e7ec16 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sun, 20 Sep 2026 15:17:04 -0400 Subject: [PATCH 11/16] Settle clippy on the fabro-petri refactors Four doc comments the projection split cut in half, the scope ledger's cache miss over an Option