diff --git a/lib/apps/fabro-cli/tests/it/scenario/petri_docker.rs b/lib/apps/fabro-cli/tests/it/scenario/petri_docker.rs index 7c0e62f90..f890e51a8 100644 --- a/lib/apps/fabro-cli/tests/it/scenario/petri_docker.rs +++ b/lib/apps/fabro-cli/tests/it/scenario/petri_docker.rs @@ -281,7 +281,7 @@ fn futures_lite_block_on(future: impl std::future::Future) -> T { fn restore_actions(server: &RunningServer, run_id: &str) -> Vec { let log = std::fs::read_to_string(server.worker_log(run_id)).unwrap_or_default(); log.lines() - .filter(|line| line.contains("sandbox workspace brought to its durable snapshot")) + .filter(|line| line.contains("workspace brought to its durable snapshot")) .filter_map(|line| { line.split_whitespace() .find_map(|word| word.strip_prefix("action=").map(str::to_owned)) diff --git a/lib/apps/fabro-server/src/petri_runs.rs b/lib/apps/fabro-server/src/petri_runs.rs index ba7efbdb5..16bef7e2b 100644 --- a/lib/apps/fabro-server/src/petri_runs.rs +++ b/lib/apps/fabro-server/src/petri_runs.rs @@ -14,12 +14,13 @@ //! `RunOptions::run_key`. use std::collections::HashMap; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use fabro_db::DbPool; use fabro_petri::SqliteRunStore; use fabro_petri::petri::{Access, OwnerId, RunKey, RunLogs, RunStore as _, StoreError}; use fabro_types::RunId; +use fabro_util::sync; use tracing::debug; pub(crate) struct PetriRuns { @@ -66,7 +67,7 @@ impl PetriRuns { ) -> Result, StoreError> { let handle = self.store.open(&Self::key(&run_id), access.clone()).await?; if let Some(owner) = access.owner() { - lock(&self.handles).insert((run_id, owner.clone()), Arc::clone(&handle)); + sync::lock(&self.handles).insert((run_id, owner.clone()), Arc::clone(&handle)); } Ok(handle) } @@ -80,7 +81,7 @@ impl PetriRuns { run_id: RunId, owner: &OwnerId, ) -> Result, StoreError> { - if let Some(handle) = lock(&self.handles).get(&(run_id, owner.clone())) { + if let Some(handle) = sync::lock(&self.handles).get(&(run_id, owner.clone())) { return Ok(Arc::clone(handle)); } let holder = self.store.owner(&Self::key(&run_id)).await?; @@ -107,7 +108,7 @@ impl PetriRuns { /// Drop the handle `owner` holds on the run: the worker's own release. /// The store ends the lease when this was the owner's last handle. pub(crate) fn release(&self, run_id: RunId, owner: &OwnerId) { - let handle = lock(&self.handles).remove(&(run_id, owner.clone())); + let handle = sync::lock(&self.handles).remove(&(run_id, owner.clone())); debug!( run_id = %run_id, owner = %owner, @@ -132,7 +133,7 @@ impl PetriRuns { /// releasing does not keep the lease. pub(crate) fn worker_exited(&self, run_id: RunId) { let dropped = { - let mut handles = lock(&self.handles); + let mut handles = sync::lock(&self.handles); let owners: Vec<_> = handles .keys() .filter(|(held, _)| *held == run_id) @@ -154,10 +155,6 @@ impl PetriRuns { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[cfg(test)] mod tests { use std::collections::BTreeMap; @@ -226,14 +223,14 @@ mod tests { } fn launched_mode(&self) -> Option<&'static str> { - *lock(&self.mode) + *sync::lock(&self.mode) } } #[async_trait::async_trait] impl WorkerRuntime for HeldWorkerRuntime { async fn start(&self, spec: WorkerLaunchSpec) -> anyhow::Result { - *lock(&self.mode) = Some(spec.mode); + *sync::lock(&self.mode) = Some(spec.mode); self.running.store(true, Ordering::SeqCst); let exit = Arc::clone(&self.exit); let stderr: Pin> = Box::pin(tokio::io::empty()); diff --git a/lib/apps/fabro-server/src/server.rs b/lib/apps/fabro-server/src/server.rs index 929e26b57..05502d0f1 100644 --- a/lib/apps/fabro-server/src/server.rs +++ b/lib/apps/fabro-server/src/server.rs @@ -1201,6 +1201,19 @@ impl AppState { &self.petri_projector } + /// The status the server holds for a managed run, so a test can wait + /// for the run to settle in the server's own map (what the delete + /// precheck reads) and not only in the stored view, which can report + /// the run ended first. + #[cfg(any(test, feature = "test-support"))] + #[must_use] + pub fn test_managed_run_status(&self, run_id: &RunId) -> Option { + self.runs + .lock() + .ok() + .and_then(|runs| runs.get(run_id).map(|managed_run| managed_run.status)) + } + /// The pool the Petri view tables live in, so a test can read them. #[cfg(any(test, feature = "test-support"))] pub fn test_petri_view_pool(&self) -> DbPool { diff --git a/lib/apps/fabro-server/tests/it/scenario/petri.rs b/lib/apps/fabro-server/tests/it/scenario/petri.rs index 9e53c3381..6deba54f1 100644 --- a/lib/apps/fabro-server/tests/it/scenario/petri.rs +++ b/lib/apps/fabro-server/tests/it/scenario/petri.rs @@ -28,7 +28,7 @@ use axum::body::Body; use axum::http::{Request, StatusCode}; use fabro_petri::engine::{self, RunStatus}; use fabro_petri::petri::{Access, OwnerId, RunKey, RunStore as _}; -use fabro_petri::{SqliteRunStore, projector}; +use fabro_petri::{SqliteRunStore, test_support}; use fabro_server::server::AppState; use fabro_server::test_support::{ TestAppStateBuilder, llm_overlay_with_provider_base_url, test_app_db_pool, @@ -237,7 +237,7 @@ pub(super) async fn settled_state( /// How many items the run's projected stream holds. async fn petri_stream_len(state: &AppState, run_id: &str) -> usize { let id: RunId = run_id.parse().expect("the run id parses"); - projector::stored_stream(&state.test_petri_view_pool(), id) + test_support::stored_stream(&state.test_petri_view_pool(), id) .await .expect("the stream reads") .len() @@ -802,6 +802,9 @@ async fn deleting_a_run_prunes_its_host_workspace_through_petri() { let store = state.test_petri_run_store(); let key = RunKey::new(run_id.clone()); wait_for_free_lease(store, &key).await; + // The view reports the run ended from Petri's own finish, before the + // server settles the managed run the delete precheck reads. + wait_for_managed_settle(&state, &run_id).await; // A live handle on the run, as its worker holds one, refuses the // delete: Petri will not prune under a lease someone holds. @@ -858,6 +861,21 @@ async fn deleting_a_run_prunes_its_host_workspace_through_petri() { } /// Wait until no owner holds the run's lease. +/// Wait until the server's own map holds the run as ended. +async fn wait_for_managed_settle(state: &AppState, run_id: &str) { + let run_id: RunId = run_id.parse().expect("a run id"); + for _ in 0..500 { + if state + .test_managed_run_status(&run_id) + .is_none_or(fabro_types::RunStatus::is_terminal) + { + return; + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + panic!("the managed run did not settle"); +} + async fn wait_for_free_lease(store: &SqliteRunStore, key: &RunKey) { for _ in 0..500 { if store.owner(key).await.expect("reads the lease").is_none() { diff --git a/lib/components/fabro-petri/src/checkpoint.rs b/lib/components/fabro-petri/src/checkpoint.rs index 48e360d1b..85a186f0d 100644 --- a/lib/components/fabro-petri/src/checkpoint.rs +++ b/lib/components/fabro-petri/src/checkpoint.rs @@ -43,8 +43,8 @@ use std::time::Duration; use fabro_checkpoint::author::GitAuthor; use fabro_checkpoint::trailer::{self, Trailer}; use fabro_store::platform_records::{DecisionRef, OperationKey}; -use fabro_types::DiffSummary; -use fabro_types::settings::run::RunCheckpointSettings; +use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace}; +use fabro_types::{DiffSummary, GitIdentitySource, SandboxProviderKind}; use petri_runtime::executor::{EnvError, ExecEnv, OutputMode, ProcessSpec, Sig}; use petri_runtime::ir::LogStream; use tokio::process::Command; @@ -92,6 +92,54 @@ pub const EXCLUDE_DIRS: &[&str] = &[ ".pytest_cache", ]; +/// The settings a run's Git work runs under, as its namespace gives them: +/// who authors the checkpoint commits and where that identity came from, +/// the checkpoint settings, and whether the sandbox provider keeps the +/// workspaces on this host. The hooks and recovery both start from it. +#[derive(Clone, Debug)] +pub struct RunGitSettings { + pub author: GitAuthor, + pub identity_source: GitIdentitySource, + pub checkpoint: RunCheckpointSettings, + /// Whether the run's workspaces are on this host (the local sandbox + /// provider). A run elsewhere snapshots inside its sandboxes. + pub host_workspaces: bool, +} + +impl From<&RunNamespace> for RunGitSettings { + fn from(settings: &RunNamespace) -> Self { + let author = settings + .git + .author + .as_ref() + .map(GitAuthor::from) + .unwrap_or_default(); + let identity_source = if author.is_default() { + GitIdentitySource::Default + } else { + GitIdentitySource::Explicit + }; + Self { + author, + identity_source, + checkpoint: settings.checkpoint.clone(), + host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, + } + } +} + +impl Default for RunGitSettings { + /// Fabro's default author and checkpoint settings, on this host. + fn default() -> Self { + Self { + author: GitAuthor::default(), + identity_source: GitIdentitySource::Default, + checkpoint: RunCheckpointSettings::default(), + host_workspaces: true, + } + } +} + /// The identity of one snapshot: the attempt whose files it holds. #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] pub struct CheckpointKey { @@ -178,6 +226,16 @@ impl CheckpointKey { } } +impl std::fmt::Display for CheckpointKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "execution {} firing {} attempt {}", + self.execution, self.firing, self.attempt + ) + } +} + /// Why a snapshot could not be taken, found or restored. #[derive(Debug, thiserror::Error)] pub enum CheckpointError { @@ -340,53 +398,19 @@ impl RunWorkspaces { .unwrap_or(false) } - /// The host site of a workspace. - fn host(&self, workspace: &str) -> Site { + /// The host site of a workspace: where `git` runs for a workspace kept + /// on this host. + #[must_use] + pub fn host(&self, workspace: &str) -> Site { Site::Host(self.workspace_path(workspace)) } /// Commit the workspace's files on the run branch as the snapshot of /// `key`, and publish it. An earlier commit of the same key that the - /// workspace still sits on, unchanged, is reused. + /// workspace still sits on, unchanged, is reused. On a sandbox site + /// `git` runs in the scope through its environment, and the commit + /// reaches the snapshot repository as a bundle. pub async fn commit( - &self, - workspace: &str, - key: CheckpointKey, - node: &str, - status: &str, - ) -> Result { - if !self.workspace_exists(workspace).await { - return Err(CheckpointError::WorkspaceMissing { - workspace: workspace.to_string(), - path: self.workspace_path(workspace), - }); - } - self.commit_at(&self.host(workspace), workspace, key, node, status) - .await - } - - /// [`commit`](Self::commit) for a workspace inside a sandbox: `git` - /// runs in the scope through `env`, and the commit reaches the - /// snapshot repository as a bundle. - pub async fn commit_in( - &self, - env: &Arc, - workspace: &str, - key: CheckpointKey, - node: &str, - status: &str, - ) -> Result { - self.commit_at( - &Site::Sandbox(Arc::clone(env)), - workspace, - key, - node, - status, - ) - .await - } - - async fn commit_at( &self, site: &Site, workspace: &str, @@ -394,6 +418,14 @@ impl RunWorkspaces { node: &str, status: &str, ) -> Result { + if let Site::Host(path) = site { + if !fs::try_exists(path).await.unwrap_or(false) { + return Err(CheckpointError::WorkspaceMissing { + workspace: workspace.to_string(), + path: path.clone(), + }); + } + } let branched = self.ensure_repository(site).await?; if let Some(existing) = self.published_sha(workspace, key).await? { if self.head(site).await?.as_deref() == Some(existing.as_str()) @@ -575,59 +607,20 @@ impl RunWorkspaces { .is_some()) } - /// The workspace's `HEAD`, or `None` when it has no commit. - pub async fn workspace_head(&self, workspace: &str) -> Result, CheckpointError> { - self.head(&self.host(workspace)).await - } - - /// [`workspace_head`](Self::workspace_head) for a workspace inside a - /// sandbox. - pub async fn workspace_head_in( - &self, - env: &Arc, - ) -> Result, CheckpointError> { - self.head(&Site::Sandbox(Arc::clone(env))).await - } - /// Whether the workspace sits on `sha` with nothing changed since. - pub async fn matches(&self, workspace: &str, sha: &str) -> Result { - self.matches_at(&self.host(workspace), sha).await - } - - /// [`matches`](Self::matches) for a workspace inside a sandbox. - pub async fn matches_in( - &self, - env: &Arc, - sha: &str, - ) -> Result { - self.matches_at(&Site::Sandbox(Arc::clone(env)), sha).await - } - - async fn matches_at(&self, site: &Site, sha: &str) -> Result { + pub async fn matches(&self, site: &Site, sha: &str) -> Result { Ok(self.head(site).await?.as_deref() == Some(sha) && self.is_clean(site).await?) } - /// Whether the host workspace's repository holds the commit `sha`, so a - /// reset can reach it; a directory that is no repository holds none. - pub async fn has_commit(&self, workspace: &str, sha: &str) -> Result { - if !self.workspace_exists(workspace).await { - return Ok(false); + /// Whether the workspace's repository holds the commit `sha`, so a + /// reset can reach it (without a transfer, in a sandbox); a directory + /// that is gone or is no repository holds none. + pub async fn has_commit(&self, site: &Site, sha: &str) -> Result { + if let Site::Host(path) = site { + if !fs::try_exists(path).await.unwrap_or(false) { + return Ok(false); + } } - self.has_commit_at(&self.host(workspace), sha).await - } - - /// Whether a sandbox workspace's repository holds the commit `sha`, so - /// a reset can reach it without a transfer. - pub async fn has_commit_in( - &self, - env: &Arc, - sha: &str, - ) -> Result { - self.has_commit_at(&Site::Sandbox(Arc::clone(env)), sha) - .await - } - - async fn has_commit_at(&self, site: &Site, sha: &str) -> Result { if self .git_status(site, "rev-parse", &["rev-parse", "--git-dir"]) .await? @@ -647,16 +640,7 @@ impl RunWorkspaces { /// Bring the workspace back to `sha`: tracked files reset, untracked /// files removed, the excluded caches left alone. - pub async fn reset(&self, workspace: &str, sha: &str) -> Result<(), CheckpointError> { - self.reset_at(&self.host(workspace), sha).await - } - - /// [`reset`](Self::reset) for a workspace inside a sandbox. - pub async fn reset_in(&self, env: &Arc, sha: &str) -> Result<(), CheckpointError> { - self.reset_at(&Site::Sandbox(Arc::clone(env)), sha).await - } - - async fn reset_at(&self, site: &Site, sha: &str) -> Result<(), CheckpointError> { + pub async fn reset(&self, site: &Site, sha: &str) -> Result<(), CheckpointError> { self.git(site, "reset", &["reset", "-q", "--hard", sha]) .await?; let mut clean = vec!["clean".to_string(), "-fdq".to_string()]; @@ -673,21 +657,38 @@ impl RunWorkspaces { } /// Recreate a gone workspace from the published snapshot `key`, at - /// `sha`, on the run branch. + /// `sha`, on the run branch. On a host site the workspace directory is + /// created and the snapshot fetched from the repository beside it. Into + /// a sandbox the snapshot enters the scope as a bundle of the + /// checkpoint's ref, and the workspace, fresh or stale, is fetched from + /// it and forced onto the run branch at `sha`. pub async fn restore( &self, + site: &Site, workspace: &str, key: CheckpointKey, sha: &str, ) -> Result<(), CheckpointError> { - let path = self.workspace_path(workspace); - fs::create_dir_all(&path) + match site { + Site::Host(path) => self.restore_on_host(path, workspace, key, sha).await, + Site::Sandbox(env) => self.restore_in_sandbox(env, workspace, key, sha).await, + } + } + + async fn restore_on_host( + &self, + path: &Path, + workspace: &str, + key: CheckpointKey, + sha: &str, + ) -> Result<(), CheckpointError> { + fs::create_dir_all(path) .await .map_err(|source| CheckpointError::Io { - path: path.clone(), + path: path.to_path_buf(), source, })?; - let site = Site::Host(path); + let site = Site::Host(path.to_path_buf()); self.git(&site, "init", &["init", "-q"]).await?; let repository = self.snapshot_repository(workspace); let repository = repository.to_string_lossy().into_owned(); @@ -710,10 +711,7 @@ impl RunWorkspaces { self.verify_restored(&site, sha).await } - /// [`restore`](Self::restore) into a sandbox: the snapshot enters the - /// scope as a bundle of the checkpoint's ref, and the workspace, fresh - /// or stale, is fetched from it and forced onto the run branch at `sha`. - pub async fn restore_in( + async fn restore_in_sandbox( &self, env: &Arc, workspace: &str, @@ -766,7 +764,7 @@ impl RunWorkspaces { "FETCH_HEAD", ]) .await?; - self.reset_at(&site, "HEAD").await?; + self.reset(&site, "HEAD").await?; self.verify_restored(&site, sha).await } .await; @@ -1083,7 +1081,8 @@ impl RunWorkspaces { .await } - async fn head(&self, site: &Site) -> Result, CheckpointError> { + /// The workspace's `HEAD`, or `None` when it has no commit. + pub async fn head(&self, site: &Site) -> Result, CheckpointError> { self.git_status(site, "rev-parse", &["rev-parse", "-q", "--verify", "HEAD"]) .await } @@ -1360,7 +1359,13 @@ mod tests { }; let first = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect("the commit"); assert!(!first.reused); @@ -1370,7 +1375,13 @@ mod tests { "the first commit created the run branch in a fresh repository" ); let again = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect("the second commit"); assert_eq!(again, Snapshot { @@ -1408,7 +1419,7 @@ mod tests { ); assert!( workspaces - .matches(workspace, &first.sha) + .matches(&workspaces.host(workspace), &first.sha) .await .expect("matches") ); @@ -1422,12 +1433,12 @@ mod tests { .expect("an untracked file"); assert!( !workspaces - .matches(workspace, &first.sha) + .matches(&workspaces.host(workspace), &first.sha) .await .expect("matches") ); workspaces - .reset(workspace, &first.sha) + .reset(&workspaces.host(workspace), &first.sha) .await .expect("the reset"); assert_eq!( @@ -1449,7 +1460,7 @@ mod tests { "the snapshot repository still knows the commit" ); workspaces - .restore(workspace, key, &first.sha) + .restore(&workspaces.host(workspace), workspace, key, &first.sha) .await .expect("the restore"); assert_eq!( @@ -1460,7 +1471,7 @@ mod tests { ); assert_eq!( workspaces - .workspace_head(workspace) + .head(&workspaces.host(workspace)) .await .expect("the head"), Some(first.sha) @@ -1483,7 +1494,13 @@ mod tests { attempt: 1, }; let error = workspaces - .commit(workspace, key, "build", "success") + .commit( + &workspaces.host(workspace), + workspace, + key, + "build", + "success", + ) .await .expect_err("the commit fails"); assert!(matches!(error, CheckpointError::Command { .. }), "{error}"); @@ -1537,7 +1554,13 @@ mod tests { attempt: 1, }; let first = workspaces - .commit(workspace, first_key, "start", "success") + .commit( + &workspaces.host(workspace), + workspace, + first_key, + "start", + "success", + ) .await .expect("the first commit"); assert_eq!( @@ -1555,7 +1578,13 @@ mod tests { attempt: 1, }; let second = workspaces - .commit(workspace, second_key, "write", "success") + .commit( + &workspaces.host(workspace), + workspace, + second_key, + "write", + "success", + ) .await .expect("the second commit"); assert_eq!(second.branched, None); diff --git a/lib/components/fabro-petri/src/fork.rs b/lib/components/fabro-petri/src/fork.rs index 70f76dd95..764777029 100644 --- a/lib/components/fabro-petri/src/fork.rs +++ b/lib/components/fabro-petri/src/fork.rs @@ -509,16 +509,12 @@ pub async fn stage_labels(views: &DbPool, run_id: RunId) -> Result), + #[error("the run has no blob table to collect artifacts into")] + NoBlobs, + #[error("the workspace could not be listed below `{root}`")] + List { + root: String, + #[source] + source: EnvError, + }, + #[error("the run's restore plan could not be read")] + Plan(#[source] RecoveryError), + #[error("{reason}")] + Unresumable { reason: String }, + #[error(transparent)] + Restore(RecoveryError), +} + +impl HookError { + /// The error and its causes on one line, for the boundaries that carry + /// text. + #[must_use] + pub fn render(&self) -> String { + collect_chain(self).join(": ") + } +} + /// What Fabro's hooks need beside the run: where the platform records go, -/// who authors the commits, the checkpoint settings, and which files are -/// the run's artifacts. +/// the run's Git settings, and which files are the run's artifacts. pub struct HooksSpec { - pub records: Arc, - pub author: GitAuthor, - /// Where the author identity came from: the run's settings, or Fabro's - /// default. - pub identity_source: GitIdentitySource, - pub checkpoint: RunCheckpointSettings, + pub records: Arc, + pub git: RunGitSettings, /// The `[run.artifacts] include` patterns: which files of a stage's /// workspace are collected after the stage. - pub artifacts: Vec, - /// Whether the run's workspaces are on this host (the local sandbox - /// provider). A run elsewhere snapshots inside its sandboxes. - pub host_workspaces: bool, + pub artifacts: Vec, /// A test's gate directory: a checkpoint point named by a `.hold` file /// there waits for its `.release` file. `None` outside tests. - pub test_gates: Option, + pub test_gates: Option, } impl HooksSpec { - /// The spec a run's settings give: its Git author, its checkpoint - /// settings, its artifact patterns, and whether its sandbox provider - /// keeps workspaces on this host. + /// The spec a run's settings give: its Git settings and its artifact + /// patterns. #[must_use] pub fn for_run(records: Arc, settings: &RunNamespace) -> Self { - let author = settings - .git - .author - .as_ref() - .map(GitAuthor::from) - .unwrap_or_default(); - let identity_source = if author.is_default() { - GitIdentitySource::Default - } else { - GitIdentitySource::Explicit - }; Self { records, - author, - identity_source, - checkpoint: settings.checkpoint.clone(), + git: RunGitSettings::from(settings), artifacts: settings.artifacts.include.clone(), - host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, test_gates: None, } } @@ -192,61 +250,145 @@ type AcquiredEnv = (String, Arc); /// The identity of a collected file: its path and content digest. type ArtifactIdentity = (String, String); -/// The last checkpoint recorded: its workspace and commit. -type LastCheckpoint = (String, String); +/// A checkpoint's workspace and commit. +type WorkspaceCommit = (String, String); -/// Fabro's `ExecutionHooks`, around the hooks the runtime installed. -pub struct FabroHooks { - inner: Arc, - run_id: RunId, - records: Arc, - /// Where an artifact's bytes and a diff's patch go; `None` records - /// summaries alone. - blobs: Option>, - workspaces: RunWorkspaces, - lookup: WorkspaceLookup, - identity: GitIdentity, - artifact_globs: Result, - host_workspaces: bool, - test_gates: Option, - handle: OnceLock, +/// The run's checkpoints as the hooks track them: what this process +/// committed, what has its platform record, which was recorded last, and +/// the run branch. +#[derive(Default)] +struct CheckpointLedger { /// The workspace and commit of every checkpoint this process made. - committed: Mutex>, - /// Which checkpoints have their platform record, loaded from the store - /// once and kept up to date with every append. - recorded: Mutex>, - recorded_loaded: OnceCell<()>, + committed: Mutex>, + /// Which checkpoints have their platform record: read from the store + /// once, then kept current with every append. + recorded: OnceCell>>, /// The workspace and commit of the checkpoint recorded last: the head /// the run's diff is measured to. - last_checkpoint: Mutex>, + last: Mutex>, /// The run branch as recorded, once: read from the store, or written /// by the commit that created the branch. - branch: OnceCell, - /// Every artifact collected so far, by path and digest, loaded from the - /// store once and kept up to date with every append. - collected: Mutex>, - collected_loaded: OnceCell<()>, - /// Inherited workspaces resolved through the run's records. - inherited: Mutex>>, - /// One lock per workspace: the branches of a parallel node and a nested - /// invocation share their caller's workspace, and Git allows one index - /// operation at a time in it. - workspace_locks: Mutex>>>, - /// The checkpoint failure that ended the run, when one did. - failure: Mutex>, + branch: OnceCell, +} + +impl CheckpointLedger { + fn remember_commit(&self, key: CheckpointKey, workspace: &str, sha: &str) { + sync::lock(&self.committed).insert(key, (workspace.to_string(), sha.to_string())); + } + + fn commit_of(&self, key: CheckpointKey) -> Option { + sync::lock(&self.committed).get(&key).cloned() + } + + fn last(&self) -> Option { + sync::lock(&self.last).clone() + } + + fn set_last(&self, last: WorkspaceCommit) { + *sync::lock(&self.last) = Some(last); + } + + /// The last checkpoint the store named, kept only when this process + /// has not recorded one yet. + fn set_last_if_unset(&self, last: Option) { + let mut current = sync::lock(&self.last); + if current.is_none() { + *current = last; + } + } +} + +/// The run's artifacts as the hooks track them: which files are collected, +/// and which were collected already. +struct ArtifactLedger { + /// The `[run.artifacts] include` patterns, or why they do not parse. + globs: Result>, + /// Every artifact collected so far, by path and digest: read from the + /// store once, then kept current with every append. + collected: OnceCell>>, +} + +/// Where each acquired scope's workspace is, and the locks that serialize +/// Git work in it. +#[derive(Default)] +struct ScopeEnvs { /// The environment of every acquired scope, by execution and scope, /// with the workspace id the executor named: where `git` runs when the /// workspaces are not on this host, and where artifacts are read from /// on every provider. Dropped at release. - envs: Mutex>, + acquired: Mutex>, + /// Inherited workspaces resolved through the run's records. + inherited: Mutex>>, + /// One lock per workspace: the branches of a parallel node and a nested + /// invocation share their caller's workspace, and Git allows one index + /// operation at a time in it. + locks: Mutex>>>, +} + +impl ScopeEnvs { + fn insert(&self, execution: ExecutionId, scope: ScopeId, env: AcquiredEnv) { + sync::lock(&self.acquired).insert((execution, scope), env); + } + + fn remove(&self, execution: ExecutionId, scope: ScopeId) { + sync::lock(&self.acquired).remove(&(execution, scope)); + } + + fn get(&self, execution: ExecutionId, scope: ScopeId) -> Option { + sync::lock(&self.acquired).get(&(execution, scope)).cloned() + } + + /// The inherited workspace of an invocation, once resolved: `None` + /// when not resolved yet, `Some(None)` when it inherits none. + #[expect( + clippy::option_option, + reason = "the outer option is the cache miss; the inner is an invocation that inherits no workspace" + )] + fn inherited(&self, invocation: InvocationId) -> Option> { + sync::lock(&self.inherited).get(&invocation).cloned() + } + + fn remember_inherited(&self, invocation: InvocationId, inherited: Option) { + sync::lock(&self.inherited).insert(invocation, inherited); + } + + /// The lock that serializes Git work in one workspace. + fn lock_for(&self, workspace: &str) -> Arc> { + Arc::clone( + sync::lock(&self.locks) + .entry(workspace.to_string()) + .or_default(), + ) + } +} + +/// Fabro's `ExecutionHooks`, around the hooks the runtime installed. +pub struct FabroHooks { + inner: Arc, + run_id: RunId, + records: Arc, + /// Where an artifact's bytes and a diff's patch go; `None` records + /// summaries alone. + blobs: Option>, + workspaces: RunWorkspaces, + lookup: WorkspaceLookup, + identity: GitIdentity, + host_workspaces: bool, + test_gates: Option, + handle: OnceLock, + checkpoints: CheckpointLedger, + artifacts: ArtifactLedger, + scopes: ScopeEnvs, + /// The checkpoint failure that ended the run, when one did. + failure: Mutex>, /// Whether the run continues from its records: a sandbox workspace is /// then brought to its snapshot when its scope is first acquired. - resumed: bool, + resumed: bool, /// The snapshot every live sandbox workspace must sit on before work /// resumes in it, read once from the records; an entry leaves when it /// is applied. - restore: OnceCell>>, - store: Arc, + restore: OnceCell>>, + store: Arc, } impl FabroHooks { @@ -268,12 +410,16 @@ impl FabroHooks { blobs: Option>, ) -> Self { let identity = GitIdentity { - name: spec.author.name.clone(), - email: spec.author.email.clone(), - source: spec.identity_source, + name: spec.git.author.name.clone(), + email: spec.git.author.email.clone(), + source: spec.git.identity_source, }; - let workspaces = - RunWorkspaces::new(run_dir, run_id.to_string(), spec.author, &spec.checkpoint); + let workspaces = RunWorkspaces::new( + run_dir, + run_id.to_string(), + spec.git.author, + &spec.git.checkpoint, + ); Self { inner, run_id, @@ -282,21 +428,16 @@ impl FabroHooks { workspaces, lookup: WorkspaceLookup::new(Arc::clone(&store), run_key), identity, - artifact_globs: WorkspaceGlobSet::try_new(&spec.artifacts), - host_workspaces: spec.host_workspaces, + host_workspaces: spec.git.host_workspaces, test_gates: spec.test_gates, handle: OnceLock::new(), - committed: Mutex::default(), - recorded: Mutex::default(), - recorded_loaded: OnceCell::new(), - last_checkpoint: Mutex::default(), - branch: OnceCell::new(), - collected: Mutex::default(), - collected_loaded: OnceCell::new(), - inherited: Mutex::default(), - workspace_locks: Mutex::default(), + checkpoints: CheckpointLedger::default(), + artifacts: ArtifactLedger { + globs: WorkspaceGlobSet::try_new(&spec.artifacts).map_err(Arc::new), + collected: OnceCell::new(), + }, + scopes: ScopeEnvs::default(), failure: Mutex::default(), - envs: Mutex::default(), resumed, restore: OnceCell::new(), store, @@ -315,7 +456,7 @@ impl FabroHooks { /// engine reports the run failed with. #[must_use] pub fn checkpoint_failure(&self) -> Option { - lock(&self.failure).clone() + sync::lock(&self.failure).clone() } /// The run's workspaces on this host, as the hooks reach them. @@ -324,17 +465,8 @@ impl FabroHooks { &self.workspaces } - /// The lock that serializes Git work in one workspace. - fn workspace_lock(&self, workspace: &str) -> Arc> { - Arc::clone( - lock(&self.workspace_locks) - .entry(workspace.to_string()) - .or_default(), - ) - } - fn fail_run(&self, message: &str) { - let mut failure = lock(&self.failure); + let mut failure = sync::lock(&self.failure); if failure.is_none() { *failure = Some(message.to_string()); } @@ -353,27 +485,29 @@ impl FabroHooks { /// The workspace id of `scope` in the context's invocation: the /// isolated name when its workspace exists, else the inherited one the /// records name, else the isolated name for the caller to report. - async fn workspace_of(&self, context: &HookContext, scope: ScopeId) -> Result { + async fn workspace_of( + &self, + context: &HookContext, + scope: ScopeId, + ) -> Result { let isolated = workspace::isolated_workspace(context.invocation, scope); if self.workspaces.workspace_exists(&isolated).await { return Ok(isolated); } - let cached = lock(&self.inherited).get(&context.invocation).cloned(); - let inherited = if let Some(inherited) = cached { + let inherited = if let Some(inherited) = self.scopes.inherited(context.invocation) { inherited } else { let inherited = self .lookup .inherited(context.invocation) .await - .map_err(|error| { - format!( - "the workspace of scope {scope} in invocation {} could not be found: {}", - context.invocation, - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Lookup { + scope, + invocation: context.invocation, + source, })?; - lock(&self.inherited).insert(context.invocation, inherited.clone()); + self.scopes + .remember_inherited(context.invocation, inherited.clone()); inherited }; Ok(inherited.unwrap_or(isolated)) @@ -382,12 +516,34 @@ impl FabroHooks { /// The environment of `scope` in the context's execution, as /// `scope_acquired` kept it, with the workspace id the executor named. fn env_of(&self, context: &HookContext, scope: ScopeId) -> Option { - lock(&self.envs).get(&(context.execution, scope)).cloned() + self.scopes.get(context.execution, scope) + } + + /// Where `scope`'s workspace is and what to call it: on this host, the + /// directory the records name (`None` when it does not exist yet); in + /// a sandbox, the environment kept at `scope_acquired` (`None` before + /// the scope was acquired). + async fn site_of( + &self, + context: &HookContext, + scope: ScopeId, + ) -> Result, HookError> { + if !self.host_workspaces { + return Ok(self + .env_of(context, scope) + .map(|(workspace, env)| (workspace, Site::Sandbox(env)))); + } + let workspace = self.workspace_of(context, scope).await?; + if !self.workspaces.workspace_exists(&workspace).await { + return Ok(None); + } + let site = self.workspaces.host(&workspace); + Ok(Some((workspace, site))) } /// The checkpoint commit for one attempt's result. `Ok(Some)` is the /// note to record, `Ok(None)` nothing to record, `Err` the fatal - /// failure message. + /// failure. async fn snapshot( &self, context: &HookContext, @@ -396,16 +552,10 @@ impl FabroHooks { node: &str, status: &Status, origin: ResultOrigin, - ) -> Result, String> { - if !self.host_workspaces { - return self - .snapshot_in_sandbox(context, scope, key, node, status, origin) - .await; - } - let workspace = self.workspace_of(context, scope).await?; - if !self.workspaces.workspace_exists(&workspace).await { + ) -> Result, HookError> { + let Some((workspace, site)) = self.site_of(context, scope).await? else { // A skipped node or a driver-made outcome may precede the scope's - // environment; nothing of the stage's is on disk to snapshot. + // environment; nothing of the stage's exists to snapshot. if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) { return Ok(Some(Note::new( CHECKPOINT_NOTE, @@ -413,22 +563,21 @@ impl FabroHooks { "execution": key.execution, "firing": key.firing, "attempt": key.attempt, - "workspace": workspace, - "skipped": "the workspace does not exist yet", + "skipped": "the scope has no workspace yet", }), ))); } - return Err(format!( - "the workspace `{workspace}` of scope {scope} does not exist at {}", - self.workspaces.workspace_path(&workspace).display() - )); - } + return Err(HookError::NoWorkspace { + scope, + execution: context.execution, + }); + }; self.gate("commit", node).await; - let serialized = self.workspace_lock(&workspace); + let serialized = self.scopes.lock_for(&workspace); let _held = serialized.lock().await; match self .workspaces - .commit(&workspace, key, node, status.tag()) + .commit(&site, &workspace, key, node, status.tag()) .await { Ok(snapshot) => { @@ -439,6 +588,7 @@ impl FabroHooks { firing = key.firing, attempt = key.attempt, reused = snapshot.reused, + site = ?site, "checkpoint committed" ); self.committed(key, &workspace, &snapshot).await; @@ -454,85 +604,18 @@ impl FabroHooks { }), ))) } - Err(error) => Err(format!( - "the checkpoint commit of `{node}` failed: {}", - collect_chain(&error).join(": ") - )), - } - } - - /// [`snapshot`](Self::snapshot) for a workspace inside the scope's - /// sandbox, through the environment kept at `scope_acquired`. - async fn snapshot_in_sandbox( - &self, - context: &HookContext, - scope: ScopeId, - key: CheckpointKey, - node: &str, - status: &Status, - origin: ResultOrigin, - ) -> Result, String> { - let Some((workspace, env)) = self.env_of(context, scope) else { - // A skipped node or a driver-made outcome may precede the scope's - // environment; nothing of the stage's exists to snapshot. - if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) { - return Ok(Some(Note::new( - CHECKPOINT_NOTE, - json!({ - "execution": key.execution, - "firing": key.firing, - "attempt": key.attempt, - "skipped": "the scope has no environment yet", - }), - ))); - } - return Err(format!( - "scope {scope} of execution {} has no sandbox environment to snapshot in", - context.execution - )); - }; - self.gate("commit", node).await; - let serialized = self.workspace_lock(&workspace); - let _held = serialized.lock().await; - match self - .workspaces - .commit_in(&env, &workspace, key, node, status.tag()) - .await - { - Ok(snapshot) => { - debug!( - run_id = %self.run_id, - node, - execution = key.execution, - firing = key.firing, - attempt = key.attempt, - reused = snapshot.reused, - "checkpoint committed in the sandbox" - ); - self.committed(key, &workspace, &snapshot).await; - Ok(Some(Note::new( - CHECKPOINT_NOTE, - json!({ - "execution": key.execution, - "firing": key.firing, - "attempt": key.attempt, - "workspace": workspace, - "git_commit_sha": snapshot.sha, - "reused": snapshot.reused, - }), - ))) - } - Err(error) => Err(format!( - "the checkpoint commit of `{node}` in the sandbox failed: {}", - collect_chain(&error).join(": ") - )), + Err(source) => Err(HookError::Commit { + node: node.to_string(), + source, + }), } } /// Remember a commit this process made, and record the run branch when /// this commit created it. async fn committed(&self, key: CheckpointKey, workspace: &str, snapshot: &Snapshot) { - lock(&self.committed).insert(key, (workspace.to_string(), snapshot.sha.clone())); + self.checkpoints + .remember_commit(key, workspace, &snapshot.sha); let Some(branched) = &snapshot.branched else { return; }; @@ -543,7 +626,7 @@ impl FabroHooks { .clone() .unwrap_or_else(|| snapshot.sha.clone()); if let Err(error) = self.record_branch(key, workspace, base_sha).await { - warn!(run_id = %self.run_id, error = %error, "the run branch was not recorded"); + warn!(run_id = %self.run_id, error = %error.render(), "the run branch was not recorded"); } } @@ -560,16 +643,17 @@ impl FabroHooks { key: CheckpointKey, workspace: &str, base_sha: String, - ) -> Result<(), String> { + ) -> Result<(), HookError> { let position = StagePosition { execution: key.execution, firing: key.firing, }; let branch = self + .checkpoints .branch .get_or_try_init(|| async { if let Some(stored) = self.stored_branch().await? { - return Ok::<_, String>(stored); + return Ok::<_, HookError>(stored); } let record = RunBranchRecord { run_branch: Some(self.workspaces.run_branch()), @@ -583,11 +667,9 @@ impl FabroHooks { Some(position), ) .await - .map_err(|error| { - format!( - "the run branch record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "run branch", + source, })?; let identity = PlatformRecord::GitIdentity(GitIdentityRecord { identity: self.identity.clone(), @@ -595,11 +677,9 @@ impl FabroHooks { self.records .append(&self.run_id, &identity, Some(position)) .await - .map_err(|error| { - format!( - "the git identity record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "git identity", + source, })?; info!( run_id = %self.run_id, @@ -615,16 +695,14 @@ impl FabroHooks { } /// The run branch the store already holds, when a record exists. - async fn stored_branch(&self) -> Result, String> { + async fn stored_branch(&self) -> Result, HookError> { let stored = self .records .read_kind(&self.run_id, PlatformRecordKind::RunBranch) .await - .map_err(|error| { - format!( - "the run's branch record could not be read: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Read { + kind: "branch", + source, })?; Ok(stored.into_iter().find_map(|stored| match stored.record { PlatformRecord::RunBranch(record) => Some(record), @@ -634,9 +712,7 @@ impl FabroHooks { /// The restore plan of a resumed run, read once: what every live /// sandbox workspace must be brought to at its first acquisition. - async fn restore_targets( - &self, - ) -> Result<&Mutex>, ScopeAcquiredError> { + async fn restore_targets(&self) -> Result<&Mutex>, HookError> { self.restore .get_or_try_init(|| async { let plan = recovery::plan( @@ -646,80 +722,39 @@ impl FabroHooks { &self.workspaces, ) .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the run's restore plan could not be read: {}", - collect_chain(&error).join(": ") - )) - })?; + .map_err(HookError::Plan)?; match plan { Plan::Resume { targets } => Ok(Mutex::new(targets)), Plan::Start => Ok(Mutex::default()), - Plan::Failed { reason } => Err(ScopeAcquiredError::new(reason)), + Plan::Failed { reason } => Err(HookError::Unresumable { reason }), } }) .await } - /// Bring a host workspace to the snapshot the resumed run's durable - /// state names, once, at its first acquisition. After a restart the - /// server already brought it there, so this verifies; a fork's fresh - /// workspace is restored here from the snapshot repository the fork - /// seeded. - async fn restore_host(&self, workspace: &str) -> Result<(), ScopeAcquiredError> { + /// Bring a workspace to the snapshot the resumed run's durable state + /// names, once, at its first acquisition. After a restart the server + /// already brought a host workspace there, so this verifies; a fork's + /// fresh workspace is restored here from the snapshot repository the + /// fork seeded; a sandbox workspace is only reachable here. + async fn restore(&self, workspace: &str, site: &Site) -> Result<(), HookError> { let targets = self.restore_targets().await?; - let target = lock(targets).remove(workspace); + let target = sync::lock(targets).remove(workspace); let Some(target) = target else { return Ok(()); }; - let serialized = self.workspace_lock(workspace); + let serialized = self.scopes.lock_for(workspace); let _held = serialized.lock().await; - let action = recovery::bring_host_to(&self.workspaces, workspace, &target) + let action = recovery::bring_to(&self.workspaces, site, workspace, &target) .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the host workspace `{workspace}` could not be brought to its snapshot: {}", - collect_chain(&error).join(": ") - )) - })?; + .map_err(HookError::Restore)?; info!( run_id = %self.run_id, workspace, sha = target.sha, action = ?action, - "host workspace brought to its durable snapshot" - ); - Ok(()) - } - - /// Bring a sandbox workspace to the snapshot the resumed run's durable - /// state names, once, at its first acquisition. - async fn restore_sandbox( - &self, - workspace: &str, - env: &Arc, - ) -> Result<(), ScopeAcquiredError> { - let targets = self.restore_targets().await?; - let target = lock(targets).remove(workspace); - let Some(target) = target else { - return Ok(()); - }; - let serialized = self.workspace_lock(workspace); - let _held = serialized.lock().await; - let action = recovery::bring_sandbox_to(&self.workspaces, env, workspace, &target) - .await - .map_err(|error| { - ScopeAcquiredError::new(format!( - "the sandbox workspace `{workspace}` could not be brought to its snapshot: {}", - collect_chain(&error).join(": ") - )) - })?; - info!( - run_id = %self.run_id, - workspace, - sha = target.sha, - action = ?action, - "sandbox workspace brought to its durable snapshot" + site = ?site, + "workspace brought to its durable snapshot" ); Ok(()) } @@ -731,15 +766,12 @@ impl FabroHooks { context: &HookContext, scope: ScopeId, key: CheckpointKey, - ) -> Result<(), String> { - self.recorded_loaded - .get_or_try_init(|| self.load_recorded()) - .await?; - if lock(&self.recorded).contains(&key) { + ) -> Result<(), HookError> { + let recorded = self.recorded_checkpoints().await?; + if sync::lock(recorded).contains(&key) { return Ok(()); } - let committed = lock(&self.committed).get(&key).cloned(); - let (workspace, sha) = if let Some(committed) = committed { + let (workspace, sha) = if let Some(committed) = self.checkpoints.commit_of(key) { committed } else { let acquired = self.env_of(context, scope).map(|(workspace, _)| workspace); @@ -747,23 +779,13 @@ impl FabroHooks { Some(workspace) => workspace, None => self.workspace_of(context, scope).await?, }; - let serialized = self.workspace_lock(&workspace); + let serialized = self.scopes.lock_for(&workspace); let held = serialized.lock().await; let found = self.workspaces.find(&workspace, key).await; drop(held); let sha = found - .map_err(|error| { - format!( - "the checkpoint commit could not be looked up: {}", - collect_chain(&error).join(": ") - ) - })? - .ok_or_else(|| { - format!( - "no checkpoint commit exists for execution {} firing {} attempt {}", - key.execution, key.firing, key.attempt - ) - })?; + .map_err(HookError::Find)? + .ok_or(HookError::NoCommit { key })?; (workspace, sha) }; let (diff_summary, patch_blob) = match self.stage_diff(&workspace, &sha).await { @@ -774,7 +796,7 @@ impl FabroHooks { run_id = %self.run_id, workspace, sha, - error = %error, + error = %error.render(), "the checkpoint's diff was not computed" ); (None, None) @@ -800,14 +822,12 @@ impl FabroHooks { }), ) .await - .map_err(|error| { - format!( - "the checkpoint record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "checkpoint", + source, })?; - lock(&self.recorded).insert(key); - *lock(&self.last_checkpoint) = Some((workspace, sha)); + sync::lock(recorded).insert(key); + self.checkpoints.set_last((workspace, sha)); Ok(()) } @@ -819,12 +839,16 @@ impl FabroHooks { &self, workspace: &str, sha: &str, - ) -> Result<(Option, Option), String> { + ) -> Result<(Option, Option), HookError> { + let failed = |source| HookError::Diff { + what: "stage's diff", + source, + }; let parent = self .workspaces .commit_parent(workspace, sha) .await - .map_err(|error| collect_chain(&error).join(": "))?; + .map_err(failed)?; let Some(parent) = parent else { return Ok((None, None)); }; @@ -832,14 +856,14 @@ impl FabroHooks { .workspaces .diff(workspace, Some(&parent), sha) .await - .map_err(|error| collect_chain(&error).join(": "))?; + .map_err(failed)?; let patch_blob = self.patch_blob(&diff).await?; Ok((Some(diff.summary), patch_blob)) } /// The patch of a diff in the blob table, when the diff is not empty /// and the run has a blob table. - async fn patch_blob(&self, diff: &WorkspaceDiff) -> Result, String> { + async fn patch_blob(&self, diff: &WorkspaceDiff) -> Result, HookError> { if diff.is_empty() { return Ok(None); } @@ -850,49 +874,51 @@ impl FabroHooks { .write(diff.patch.as_bytes()) .await .map(Some) - .map_err(|error| format!("the patch could not be stored: {error:#}")) + .map_err(|source| HookError::Blob { + what: "patch".to_string(), + source, + }) } /// The checkpoints already recorded for the run, read once: what a /// resume's reissued routing decisions must not record again, and /// where the run's diff is measured to when this process made no /// checkpoint yet. - async fn load_recorded(&self) -> Result<(), String> { - let stored = self - .records - .read_kind(&self.run_id, PlatformRecordKind::Checkpoint) + async fn recorded_checkpoints(&self) -> Result<&Mutex>, HookError> { + self.checkpoints + .recorded + .get_or_try_init(|| async { + let stored = self + .records + .read_kind(&self.run_id, PlatformRecordKind::Checkpoint) + .await + .map_err(|source| HookError::Read { + kind: "checkpoint", + source, + })?; + let mut recorded = HashSet::new(); + let mut last = None; + for record in stored { + let PlatformRecord::Checkpoint(checkpoint) = &record.record else { + continue; + }; + if let Some(key) = checkpoint + .operation + .as_ref() + .and_then(CheckpointKey::from_operation) + { + recorded.insert(key); + } + if let (Some(workspace), Some(sha)) = + (&checkpoint.workspace, &checkpoint.git_commit_sha) + { + last = Some((workspace.clone(), sha.clone())); + } + } + self.checkpoints.set_last_if_unset(last); + Ok(Mutex::new(recorded)) + }) .await - .map_err(|error| { - format!( - "the run's checkpoint records could not be read: {}", - collect_chain(&error).join(": ") - ) - })?; - let mut recorded = lock(&self.recorded); - let mut last = None; - for record in stored { - let PlatformRecord::Checkpoint(checkpoint) = &record.record else { - continue; - }; - if let Some(key) = checkpoint - .operation - .as_ref() - .and_then(CheckpointKey::from_operation) - { - recorded.insert(key); - } - if let (Some(workspace), Some(sha)) = - (&checkpoint.workspace, &checkpoint.git_commit_sha) - { - last = Some((workspace.clone(), sha.clone())); - } - } - drop(recorded); - let mut last_checkpoint = lock(&self.last_checkpoint); - if last_checkpoint.is_none() { - *last_checkpoint = last; - } - Ok(()) } /// The artifacts of a finished attempt: every file of its workspace @@ -904,10 +930,10 @@ impl FabroHooks { context: &HookContext, scope: ScopeId, key: CheckpointKey, - ) -> Result { - let globs = match &self.artifact_globs { + ) -> Result { + let globs = match &self.artifacts.globs { Ok(globs) => globs, - Err(error) => return Err(format!("invalid run.artifacts.include pattern: {error}")), + Err(error) => return Err(HookError::Globs(Arc::clone(error))), }; if globs.is_empty() { return Ok(0); @@ -918,11 +944,9 @@ impl FabroHooks { return Ok(0); }; let Some(blobs) = &self.blobs else { - return Err("the run has no blob table to collect artifacts into".to_string()); + return Err(HookError::NoBlobs); }; - self.collected_loaded - .get_or_try_init(|| self.load_collected()) - .await?; + let already = self.collected_artifacts().await?; let candidates = list_artifacts(env.as_ref(), globs).await?; let limit = usize::try_from(ARTIFACT_MAX_FILE_BYTES).unwrap_or(usize::MAX); let mut collected = 0; @@ -941,13 +965,16 @@ impl FabroHooks { }; let digest = BlobHash::new(&bytes); let identity = (path.clone(), digest.to_string()); - if lock(&self.collected).contains(&identity) { + if sync::lock(already).contains(&identity) { continue; } let blob = blobs .write(&bytes) .await - .map_err(|error| format!("the artifact `{path}` could not be stored: {error:#}"))?; + .map_err(|source| HookError::Blob { + what: format!("artifact `{path}`"), + source, + })?; let record = PlatformRecord::ArtifactCollected(ArtifactCollectedRecord { execution: key.execution, firing: key.firing, @@ -968,13 +995,11 @@ impl FabroHooks { }), ) .await - .map_err(|error| { - format!( - "the artifact record for `{path}` could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "artifact", + source, })?; - lock(&self.collected).insert(identity); + sync::lock(already).insert(identity); total_bytes = total_bytes.saturating_add(size); collected += 1; } @@ -983,35 +1008,39 @@ impl FabroHooks { /// The artifacts already collected for the run, read once: a file that /// is unchanged since it was collected is not collected again. - async fn load_collected(&self) -> Result<(), String> { - let stored = self - .records - .read_kind(&self.run_id, PlatformRecordKind::ArtifactCollected) + async fn collected_artifacts(&self) -> Result<&Mutex>, HookError> { + self.artifacts + .collected + .get_or_try_init(|| async { + let stored = self + .records + .read_kind(&self.run_id, PlatformRecordKind::ArtifactCollected) + .await + .map_err(|source| HookError::Read { + kind: "artifact", + source, + })?; + let collected = stored + .into_iter() + .filter_map(|record| match record.record { + PlatformRecord::ArtifactCollected(artifact) => { + Some((artifact.path, artifact.digest)) + } + _ => None, + }) + .collect(); + Ok(Mutex::new(collected)) + }) .await - .map_err(|error| { - format!( - "the run's artifact records could not be read: {}", - collect_chain(&error).join(": ") - ) - })?; - let mut collected = lock(&self.collected); - for record in stored { - if let PlatformRecord::ArtifactCollected(artifact) = record.record { - collected.insert((artifact.path, artifact.digest)); - } - } - Ok(()) } /// The run's diff: the run branch's last checkpoint against the base /// the branch started from, in the snapshot repository on this host. /// Nothing is recorded for a run that never created its branch or /// never checkpointed. - async fn record_run_diff(&self) -> Result<(), String> { - self.recorded_loaded - .get_or_try_init(|| self.load_recorded()) - .await?; - let branch = match self.branch.get() { + async fn record_run_diff(&self) -> Result<(), HookError> { + self.recorded_checkpoints().await?; + let branch = match self.checkpoints.branch.get() { Some(branch) => Some(branch.clone()), None => self.stored_branch().await?, }; @@ -1022,8 +1051,7 @@ impl FabroHooks { let Some(base_sha) = branch.base_sha.clone() else { return Ok(()); }; - let last = lock(&self.last_checkpoint).clone(); - let Some((workspace, head_sha)) = last else { + let Some((workspace, head_sha)) = self.checkpoints.last() else { debug!(run_id = %self.run_id, "no checkpoint is recorded; no run diff"); return Ok(()); }; @@ -1035,11 +1063,9 @@ impl FabroHooks { .workspaces .diff(&workspace, Some(&base_sha), &head_sha) .await - .map_err(|error| { - format!( - "the run's diff could not be computed: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Diff { + what: "run's diff", + source, })?; let patch_blob = self.patch_blob(&diff).await?; let record = PlatformRecord::RunDiff(RunDiffRecord { @@ -1051,11 +1077,9 @@ impl FabroHooks { self.records .append(&self.run_id, &record, None) .await - .map_err(|error| { - format!( - "the run diff record could not be written: {}", - collect_chain(&error).join(": ") - ) + .map_err(|source| HookError::Write { + kind: "run diff", + source, })?; info!( run_id = %self.run_id, @@ -1091,7 +1115,7 @@ impl FabroHooks { async fn list_artifacts( env: &dyn ExecEnv, globs: &WorkspaceGlobSet, -) -> Result, String> { +) -> Result, HookError> { let mut files = Vec::new(); for root in globs.traversal_roots() { let listed = env @@ -1100,8 +1124,9 @@ async fn list_artifacts( ARTIFACT_LIST_DEPTH, ) .await - .map_err(|error| { - format!("the workspace could not be listed below `{root}`: {error}") + .map_err(|source| HookError::List { + root: root.to_string(), + source, })?; for entry in listed { if entry.is_dir { @@ -1149,10 +1174,6 @@ fn select_artifacts(mut candidates: Vec<(String, u64)>) -> Vec<(String, u64)> { selected } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - fn is_checkpoint_failure(status: &Status) -> bool { matches!(status, Status::Failure(info) if info.class.as_str() == CHECKPOINT_FAILED_CLASS) } @@ -1192,7 +1213,8 @@ impl ExecutionHooks for FabroHooks { { Ok(Some(note)) => prepared.notes.push(note), Ok(None) => {} - Err(message) => { + Err(error) => { + let message = error.render(); warn!( run_id = %self.run_id, node, @@ -1244,7 +1266,7 @@ impl ExecutionHooks for FabroHooks { error = %problem, "the checkpoint record was not written" ); - problems.push(problem); + problems.push(problem.render()); } match self.collect_artifacts(context, scope, key).await { Ok(0) => {} @@ -1267,7 +1289,7 @@ impl ExecutionHooks for FabroHooks { error = %problem, "artifact collection failed" ); - problems.push(format!("artifact collection failed: {problem}")); + problems.push(format!("artifact collection failed: {}", problem.render())); } } let mut report = self.inner.transition(context, transition).await?; @@ -1283,7 +1305,7 @@ impl ExecutionHooks for FabroHooks { "Petri run finished; recording the run's diff and running the run-end hooks" ); if let Err(error) = self.record_run_diff().await { - warn!(run_id = %self.run_id, error = %error, "the run's diff was not recorded"); + warn!(run_id = %self.run_id, error = %error.render(), "the run's diff was not recorded"); } self.inner.run_finished(context, finished).await } @@ -1297,7 +1319,7 @@ impl ExecutionHooks for FabroHooks { ); let scope = released.scope; let notes = self.inner.scope_released(context, released).await; - lock(&self.envs).remove(&(context.execution, scope)); + self.scopes.remove(context.execution, scope); notes } @@ -1308,18 +1330,22 @@ impl ExecutionHooks for FabroHooks { ) -> Result<(), ScopeAcquiredError> { self.inner.scope_acquired(context, acquired.clone()).await?; let workspace = acquired.workspace.as_str().to_owned(); - lock(&self.envs).insert( - (context.execution, acquired.scope), + self.scopes.insert( + context.execution, + acquired.scope, (workspace.clone(), Arc::clone(&acquired.env)), ); if !self.resumed { return Ok(()); } - if self.host_workspaces { - self.restore_host(&workspace).await + let site = if self.host_workspaces { + self.workspaces.host(&workspace) } else { - self.restore_sandbox(&workspace, &acquired.env).await - } + Site::Sandbox(Arc::clone(&acquired.env)) + }; + self.restore(&workspace, &site) + .await + .map_err(|error| ScopeAcquiredError::new(error.render())) } } diff --git a/lib/components/fabro-petri/src/http_store.rs b/lib/components/fabro-petri/src/http_store.rs index 76ee9b69c..1a08588ee 100644 --- a/lib/components/fabro-petri/src/http_store.rs +++ b/lib/components/fabro-petri/src/http_store.rs @@ -62,13 +62,14 @@ use std::collections::HashMap; use std::future::Future; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError, Weak}; +use std::sync::{Arc, Mutex, Weak}; use std::time::Duration; use std::{fmt, mem, ptr}; use fabro_api::types::{PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriRecord}; use fabro_client::{Client, api_failure_for}; use fabro_types::{BlobHash, RunId}; +use fabro_util::sync; use petri_store::{Access, Digest, LogId, OwnerId, Record, RunKey, RunLogs, RunStore, StoreError}; use serde_json::Value; use tokio::runtime::Handle; @@ -148,7 +149,7 @@ impl HttpRunStore { owner: OwnerId, locator: String, ) -> Arc { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let slot = (key.clone(), owner.clone()); if let Some(handle) = live.get(&slot).and_then(Weak::upgrade) { return handle; @@ -185,7 +186,7 @@ impl Shared { /// Await every release a dropped handle spawned, so what follows sees /// the lease as the drops left it. async fn drain_releases(&self) { - let pending = mem::take(&mut *lock(&self.releases)); + let pending = mem::take(&mut *sync::lock(&self.releases)); for release in pending { // A release task never panics: it reports its own failure. let _ = release.await; @@ -436,7 +437,7 @@ impl Drop for HttpRunLogs { return; }; { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let slot = (self.key.clone(), owner.clone()); let this: *const Self = self; if live @@ -454,7 +455,7 @@ impl Drop for HttpRunLogs { let release = runtime.spawn(async move { shared.release(&key, run_id, &owner).await; }); - lock(&self.shared.releases).push(release); + sync::lock(&self.shared.releases).push(release); } Err(_) => { warn!( @@ -551,10 +552,6 @@ impl RunLogs for HttpRunLogs { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[cfg(test)] mod tests { use fabro_client::ApiFailure; diff --git a/lib/components/fabro-petri/src/projection.rs b/lib/components/fabro-petri/src/projection.rs deleted file mode 100644 index 54183cc9f..000000000 --- a/lib/components/fabro-petri/src/projection.rs +++ /dev/null @@ -1,1834 +0,0 @@ -//! The projection of a Petri run: Petri's public events and Fabro's platform -//! records folded into the view Fabro's read side serves. -//! -//! The fold is pure. [`RunView`] holds the [`RunProjection`] the API serves -//! (`GET /runs/{id}/state`, the run list through its summary) and the -//! bookkeeping the fold needs between items ([`FoldState`]): which Petri -//! firing each stage is, which invocation each execution belongs to and -//! whether it is a parallel branch, which stage asked each open question. -//! Both halves are stored by the projector and reloaded for the next pass, -//! so a pass folds only the items past the committed positions. -//! -//! The mapping follows `VIEWS.md`, row by row. The stage key is `(execution, -//! firing)`; Fabro's `StageId` (`node@visit`) is the display label the -//! `RunProjection` keys stages by, and a label two firings would share (two -//! child invocations with the same node name and visit) is made unique by -//! naming the execution. What the matrix leaves default is left default -//! here and named in the crate's README. -//! -//! Every item the fold sees carries the delivery sequence the projector -//! assigned it (`stream_seq`), which a checkpoint keeps as its `seq`. A -//! stage's `first_event_seq`, the key the stage list sorts by, is not the -//! delivery sequence: two logs' records can be committed in an order that -//! differs from their recording times by a few positions, and the view -//! built live must equal the view rebuilt from the records alone. It is the -//! milliseconds from the run's creation to the stage's `visit.started`, -//! plus one, which is the same however the records were delivered. - -use std::collections::{BTreeMap, BTreeSet}; - -use chrono::{DateTime, TimeZone as _, Utc}; -use fabro_store::platform_records::{ - PlatformRecord, RunLifecycleKind, RunLifecycleRecord, StoredPlatformRecord, -}; -use fabro_types::settings::run::RunEnvironmentSettings; -use fabro_types::{ - BlockedReason, CheckpointRecord as ViewCheckpoint, CodingAgentEvent, CodingEvent, Conclusion, - FailureCategory, FailureDetail, FailureReason, InterviewOption, InterviewQuestionRecord, - ModelRef, ModelUsage, ParallelBranchId, ParallelBranchResult, PendingInterviewRecord, - PullRequestCreation, PullRequestCreationStatus, PullRequestLink, ReviewTarget, - ReviewTargetKind, RunApproval, RunApprovalState, RunArtifact, RunControlAction, RunDiff, - RunFailure, RunId, RunProjection, RunSandbox, RunSandboxFailure, RunSandboxInstance, - RunSandboxPlan, RunSandboxRuntime, RunStatus, RunTiming, SandboxProviderKind, StageCompletion, - StageHandler, StageId, StageInferenceProjection, StageModelUsage, StageOutcome, - StageProjection, StageState, StageTiming, StartRecord, SuccessReason, ToolCategory, ToolSource, - ToolSummary, first_event_seq, format_blob_ref, parse_blob_ref, timing, usage_rollup, -}; -use lithos_llm::catalog::{ModelId, ProviderId}; -use lithos_llm::types::Usage; -use petri_execution::events::{Derived, NodeRef, Parsed, RunEvent, Subject, ViewEvent, WaitState}; -use petri_execution::{CoordinatorEvent, ExecutionId, InvocationId}; -use petri_runtime::engine::{Admission, Event}; -use petri_runtime::ir::{Metrics, SandboxInstance, Status, StepEvent}; -use petri_runtime::steps::QuestionReference; -use serde::{Deserialize, Serialize}; -use serde_json::Value; -use tracing::debug; - -use crate::interview::question_type; - -/// One item the projector hands the fold, with its delivery sequence. -pub enum Item<'a> { - Petri(&'a RunEvent), - Platform(&'a StoredPlatformRecord), -} - -/// A stage as the fold knows it: its label in the projection, and what it -/// learned about it. -#[derive(Clone, Debug, Serialize, Deserialize)] -pub struct StageRef { - pub stage_id: StageId, - /// Whether the stage is a logical one the projection shows, or a - /// lowering node it keeps off the list. - pub shown: bool, - /// The node's instance name and visit, for the collision rule. - pub node_name: String, - pub visit: u32, -} - -/// What the fold knows about one invocation. -#[derive(Clone, Debug, Default, Serialize, Deserialize)] -pub struct InvocationRef { - /// The calling execution and firing, for a nested invocation. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub parent: Option<(u64, u64)>, - /// The parallel group and branch index, for a branch child. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub branch: Option<(StageId, u32)>, - /// The result the invocation recorded, for the root. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub failure: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub output: Option, -} - -/// Whether the run's durable record is whole, as the projector last read it. -#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct RecordHealth { - pub complete: bool, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub incomplete: Vec, -} - -/// The fold's bookkeeping between items. -#[derive(Clone, Debug, Default, Serialize, Deserialize)] -pub struct FoldState { - /// Stages by `":"`. - #[serde(default)] - pub stages: BTreeMap, - /// Labels taken, so a second firing with the same name and visit gets - /// its own. - #[serde(default)] - pub labels: BTreeSet, - #[serde(default)] - pub invocations: BTreeMap, - /// Which invocation each execution belongs to. - #[serde(default)] - pub executions: BTreeMap, - /// Open questions by id: the stage that asked. - #[serde(default)] - pub questions: BTreeMap, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub root: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub started_at: Option, - /// The run's recorded finish, when Petri recorded one. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub finished: Option, - /// The run branch and base sha, when they arrive before `run.started`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub run_branch: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub base_sha: Option, - #[serde(default)] - pub checkpoints: u32, - /// The run's diff as its `run.diff` record gave it, whichever side of - /// the run's finish it arrived on. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub run_diff: Option, - #[serde(default)] - pub health: RecordHealth, - /// Firings (`":"`) whose attempt has recorded a - /// finish: what a position-keyed platform record may be streamed - /// behind. - #[serde(default)] - pub finished_firings: BTreeSet, - /// Whether the run's sandbox still exists after its release - /// (`scope.released` `retained`): kept stopped, or deleted. Absent until - /// the root invocation's lease was released. The view carries the same - /// fact as `RunSandboxInstance.retained`. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub sandbox_retained: Option, -} - -impl FoldState { - /// Whether Petri recorded the run's finish. - #[must_use] - pub fn finished_run(&self) -> bool { - self.finished.is_some() - } -} - -/// The view of one run: what the API serves and what the fold keeps. -#[derive(Clone, Debug)] -pub struct RunView { - pub projection: Option, - pub state: FoldState, -} - -impl RunView { - #[must_use] - pub fn new() -> Self { - Self { - projection: None, - state: FoldState::default(), - } - } - - /// Fold one item at its delivery sequence. - pub fn fold(&mut self, item: &Item<'_>, stream_seq: u64) { - match item { - Item::Platform(record) => self.fold_platform(record, stream_seq), - Item::Petri(event) => self.fold_petri(event), - } - } - - /// The run's projection, once its `run.created` record was folded. - #[must_use] - pub fn projection(&self) -> Option<&RunProjection> { - self.projection.as_ref() - } - - // ── Platform records ──────────────────────────────────────────────── - - fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { - let at = millis(stored.recorded_at); - if let PlatformRecord::RunCreated(created) = &stored.record { - let title = created - .title - .clone() - .unwrap_or_else(|| fabro_types::infer_run_title(created.spec.graph.goal())); - let mut projection = RunProjection::new(title, created.spec.clone(), at); - projection.parent_id = created.parent_id; - projection.retried_from = created.retried_from; - projection.web_url.clone_from(&created.web_url); - projection.sandbox = Some(RunSandbox::planned(sandbox_plan( - &projection.spec.settings.run.environment, - ))); - self.projection = Some(projection); - return; - } - let Some(projection) = self.projection.as_mut() else { - debug!( - seq = stored.seq, - kind = %stored.record.kind(), - "platform record before run.created; not folded" - ); - return; - }; - touch(projection, at); - match &stored.record { - PlatformRecord::RunLifecycle(record) => fold_lifecycle(projection, record, at), - PlatformRecord::RunTitle(record) => projection.title.clone_from(&record.title), - PlatformRecord::RunParent(record) => projection.parent_id = record.parent_id, - PlatformRecord::RunArchived => projection.archived_at = Some(at), - PlatformRecord::RunUnarchived => projection.archived_at = None, - PlatformRecord::RunSuperseded(record) => { - projection.superseded_by = Some(record.new_run_id); - } - PlatformRecord::RunCreated(_) - | PlatformRecord::RunNotice(_) - | PlatformRecord::InterviewAnswered(_) - | PlatformRecord::NotificationSent(_) - | PlatformRecord::RunPaired(_) => {} - PlatformRecord::RunBranch(record) => { - self.state.run_branch.clone_from(&record.run_branch); - self.state.base_sha.clone_from(&record.base_sha); - if let Some(start) = projection.start.as_mut() { - start.run_branch.clone_from(&record.run_branch); - start.base_sha.clone_from(&record.base_sha); - } - } - PlatformRecord::GitIdentity(record) => { - projection.git_identity = Some(record.identity.clone()); - } - PlatformRecord::Checkpoint(record) => { - self.state.checkpoints = self.state.checkpoints.saturating_add(1); - let stage = self - .state - .stages - .get(&stage_key(record.execution, record.firing)); - let current_node = stage.map_or_else(String::new, |stage| stage.node_name.clone()); - let stage_id = stage - .filter(|stage| stage.shown) - .map(|stage| stage.stage_id.clone()); - let checkpoint = fabro_types::Checkpoint { - timestamp: at, - current_node: current_node.clone(), - git_commit_sha: record.git_commit_sha.clone(), - }; - // The patch stays in the blob table; the view carries its - // reference for a reader to resolve. - let patch = record.patch_blob.as_ref().map(format_blob_ref); - if let Some(stage) = stage_id.and_then(|stage_id| projection.stage_mut(&stage_id)) { - if patch.is_some() { - stage.diff.clone_from(&patch); - } - } - projection.checkpoints.push(ViewCheckpoint { - seq: u32::try_from(stream_seq).unwrap_or(u32::MAX), - checkpoint, - diff: RunDiff { - patch, - summary: record.diff_summary, - }, - }); - } - PlatformRecord::ArtifactCollected(record) => { - let stage = self - .state - .stages - .get(&stage_key(record.execution, record.firing)); - let Some(stage_id) = stage.map(|stage| stage.stage_id.clone()) else { - debug!( - seq = stored.seq, - path = record.path, - "artifact record for an unknown firing; not folded" - ); - return; - }; - projection.artifacts.push(RunArtifact { - stage_id, - retry: record.attempt, - relative_path: record.path.clone(), - size: record.bytes, - blob: record.blob, - }); - } - PlatformRecord::RunDiff(record) => { - let diff = RunDiff { - patch: record.patch_blob.as_ref().map(format_blob_ref), - summary: record.diff_summary, - }; - if let Some(conclusion) = projection.conclusion.as_mut() { - conclusion.diff = diff.clone(); - } - self.state.run_diff = Some(diff); - } - PlatformRecord::PullRequestRequested(record) => { - projection.pull_request_creation = Some(PullRequestCreation { - id: record.creation_id, - status: PullRequestCreationStatus::Pending, - model: record.model.clone(), - force: record.force, - requested_at: at, - updated_at: at, - pull_request: None, - error: None, - }); - } - PlatformRecord::PullRequestCreated(record) => { - let link = PullRequestLink { - owner: record.owner.clone(), - repo: record.repo.clone(), - number: record.number, - }; - projection.pull_request = Some(link.clone()); - if let Some(creation) = projection - .pull_request_creation - .as_mut() - .filter(|creation| creation.is_pending()) - { - creation.succeed(link, at); - } - } - PlatformRecord::PullRequestFailed(record) => { - if let Some(creation) = - projection - .pull_request_creation - .as_mut() - .filter(|creation| { - creation.is_pending() - && record - .creation_id - .is_none_or(|creation_id| creation_id == creation.id) - }) - { - creation.fail(record.error.clone(), at); - } - } - PlatformRecord::PullRequestLinked(record) => { - let link = record.link(); - projection.pull_request = Some(link.clone()); - if let Some(creation) = projection - .pull_request_creation - .as_mut() - .filter(|creation| creation.is_pending()) - { - creation.succeed(link, at); - } - } - PlatformRecord::PullRequestUnlinked(_) => { - projection.pull_request = None; - projection.pull_request_creation = None; - } - } - } - - // ── Petri events ──────────────────────────────────────────────────── - - fn fold_petri(&mut self, event: &RunEvent) { - let at = millis(event.recorded_at); - if let Some(record) = event.coordinator() { - self.fold_coordinator(record, event, at); - } else if let Some(engine) = event.engine() { - self.fold_engine(engine, event, at); - } else if let Some(view) = event.view() { - self.fold_view(view, event, at); - } - if let Some(projection) = self.projection.as_mut() { - touch(projection, at); - } - } - - fn fold_coordinator(&mut self, record: &CoordinatorEvent, event: &RunEvent, at: DateTime) { - match record { - CoordinatorEvent::RunStarted { - root, forked_from, .. - } => { - self.state.root = Some(root.raw()); - self.state.started_at = Some(event.recorded_at); - if let Some(projection) = self.projection.as_mut() { - // A fork's declaration names its source; a parse failure - // means the source was not a Fabro run, which the - // projection cannot show. - projection.forked_from = forked_from.as_ref().and_then(|origin| { - Some(fabro_types::ForkOrigin { - source_run_id: origin.source.as_str().parse().ok()?, - execution: origin.position.execution.raw(), - firing: origin.position.firing.raw(), - rerun_last: origin.rerun_last, - }) - }); - apply_status(projection, RunStatus::Running, at); - projection.start = Some(StartRecord { - start_time: at, - run_branch: self.state.run_branch.clone(), - base_sha: self.state.base_sha.clone(), - }); - // The scope's sandbox is acquired next; `scope.acquired` - // or `scope.failed` settles it. - if let Some(sandbox) = projection.sandbox.take() { - projection.sandbox = Some(RunSandbox::initializing(sandbox.plan().clone())); - } - } - } - CoordinatorEvent::InvocationDeclared { invocation, .. } => { - let mut info = InvocationRef::default(); - if let Some(parent) = &event.context.parent { - info.parent = Some((parent.execution.raw(), parent.firing.raw())); - if let Some((fork_firing, index)) = branch_slot(&parent.slot) { - let group = self - .state - .stages - .get(&stage_key(parent.execution.raw(), fork_firing)) - .map(|stage| stage.stage_id.clone()); - if let Some(group) = group { - info.branch = Some((group, index)); - } - } - } - self.state.invocations.insert(invocation.raw(), info); - } - CoordinatorEvent::ExecutionDeclared { - execution, - invocation, - .. - } => { - self.state - .executions - .insert(execution.raw(), invocation.raw()); - } - CoordinatorEvent::InvocationFinished { invocation, result } => { - let info = self.state.invocations.entry(invocation.raw()).or_default(); - info.failure = result - .failure - .as_ref() - .map(|failure| failure.message.clone()); - info.output = Some(result.output.clone()); - } - CoordinatorEvent::RunPaused => { - if let Some(projection) = self.projection.as_mut() { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } - } - } - CoordinatorEvent::RunUnpaused => { - if let Some(projection) = self.projection.as_mut() { - let next = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, next, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } - } - } - CoordinatorEvent::RunFinished { status } => { - self.state.finished = Some(status.to_string()); - self.conclude(status.to_string().as_str(), at); - } - // ── Sandbox: the retention outcome (VIEWS.md "Sandbox") ───────── - // The instance stays on `Run.sandbox`: it names what ran, and - // `retained` says whether it still exists. - CoordinatorEvent::ScopeReleased { - invocation, - retained, - .. - } => { - if Some(invocation.raw()) == self.state.root { - self.state.sandbox_retained = Some(*retained); - if let Some(sandbox) = self - .projection - .as_mut() - .and_then(|projection| projection.sandbox.as_mut()) - { - sandbox.set_retained(*retained); - } - } - } - CoordinatorEvent::GraphRegistered { .. } - | CoordinatorEvent::ExecutionFinished { .. } - | CoordinatorEvent::InvocationCancelRequested { .. } - | CoordinatorEvent::RunNoteRecorded { .. } => {} - } - } - - /// The run's conclusion, from its recorded finish and what the stages - /// summed to. - fn conclude(&mut self, status: &str, at: DateTime) { - let Some(projection) = self.projection.as_mut() else { - return; - }; - let root = self - .state - .root - .and_then(|root| self.state.invocations.get(&root)); - let failure_message = root.and_then(|root| root.failure.clone()); - let (run_status, outcome, failure) = match status { - "success" => ( - RunStatus::Succeeded { - reason: SuccessReason::Completed, - }, - StageOutcome::Succeeded, - None, - ), - "cancelled" => ( - RunStatus::Failed { - reason: FailureReason::Cancelled, - }, - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason: FailureReason::Cancelled, - detail: FailureDetail::new( - failure_message - .clone() - .unwrap_or_else(|| "the run was cancelled".to_string()), - FailureCategory::Canceled, - ), - }), - ), - _ => ( - RunStatus::Failed { - reason: FailureReason::WorkflowError, - }, - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason: FailureReason::WorkflowError, - detail: FailureDetail::new( - failure_message - .clone() - .unwrap_or_else(|| "the run failed".to_string()), - FailureCategory::Deterministic, - ), - }), - ), - }; - apply_status(projection, run_status, at); - projection.pending_control = None; - projection.pending_interviews.clear(); - let rollup = usage_rollup::usage_rollup_from_projection(projection); - let (stages, total_retries) = rollup.conclusion_stages(projection); - let wall_time_ms = self.state.started_at.map_or(0, |started| { - u64::try_from(at.timestamp_millis()) - .unwrap_or(0) - .saturating_sub(started) - }); - let timing = RunTiming::new( - wall_time_ms, - rollup.timing.inference_time_ms, - rollup.timing.tool_time_ms, - ); - let last_checkpoint = projection.checkpoints.last(); - projection.conclusion = Some(Conclusion { - timestamp: at, - status: outcome, - timing, - failure, - final_git_commit_sha: last_checkpoint - .and_then(|checkpoint| checkpoint.checkpoint.git_commit_sha.clone()), - stages, - usage: rollup.usage_if_present(), - total_retries, - diff: self - .state - .run_diff - .clone() - .or_else(|| last_checkpoint.map(|checkpoint| checkpoint.diff.clone())) - .unwrap_or_default(), - }); - } - - fn fold_engine(&mut self, engine: &Event, event: &RunEvent, at: DateTime) { - let Some(execution) = event.context.execution else { - return; - }; - match engine { - Event::AdmissionDecided { decision, .. } => { - if let Admission::Skip { outcome } = decision { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.state = StageState::Skipped; - stage.completion = Some(StageCompletion { - outcome: StageOutcome::Skipped, - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - } - } - } - Event::StepStarted { attempt, .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - if attempt.raw() > 1 { - stage.clear_live_timing(); - stage.output = None; - stage.output_bytes = None; - } - stage.state = StageState::Running; - stage.live_streaming = Some(true); - } - } - Event::StepProgressRecorded { ev, .. } => { - self.fold_progress(execution, event, ev, at); - } - Event::StepFinished { - firing, - attempt, - outcome, - } => { - self.state - .finished_firings - .insert(stage_key(execution.raw(), firing.raw())); - let is_final = matches!( - event.derived, - Some(Derived::StepFinished { is_final: true, .. }) - ); - let node_name = event - .subject - .as_ref() - .map(|subject| subject.node.name.to_string()); - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - // The step's output: a string, or a command's `stdout`, - // either of which is a `blob://` reference when the - // step offloaded it. The reference stays as it is; the - // bytes it names are the live log's. - let output = outcome - .output - .as_str() - .or_else(|| outcome.output.get("stdout").and_then(Value::as_str)); - if let Some(output) = output { - if parse_blob_ref(output).is_none() { - stage.output_bytes = Some(output.len() as u64); - } - stage.output = Some(output.to_string()); - } - // A simulated step (a dry run) answers with its text. - let simulated = outcome - .output - .get("simulated") - .and_then(Value::as_bool) - .unwrap_or(false); - if simulated - && matches!( - stage.handler, - Some(StageHandler::Prompt | StageHandler::Agent) - ) - { - if let Some(text) = outcome.output.get("text").and_then(Value::as_str) { - stage.response = Some(text.to_string()); - } - } - // An agent's answer: the `response.` the step wrote - // into the run context, as the prompt step writes it. - if stage.handler == Some(StageHandler::Agent) { - let response = node_name - .as_deref() - .and_then(|name| { - outcome - .context_updates - .get(format!("response.{name}").as_str()) - }) - .and_then(Value::as_str) - .or_else(|| outcome.output.as_str()); - if let Some(response) = response { - stage.response = Some(response.to_string()); - } - } - stage.live_streaming = Some(false); - apply_metrics(stage, &outcome.metrics); - if is_final { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - stage.termination = Some(match outcome.status { - Status::TimedOut => fabro_types::CommandTermination::TimedOut, - Status::Cancelled => fabro_types::CommandTermination::Cancelled, - Status::Success - | Status::PartialSuccess { .. } - | Status::Failure(_) - | Status::Skipped => fabro_types::CommandTermination::Exited, - }); - } else { - stage.state = StageState::Retrying; - debug!(attempt = attempt.raw(), "attempt returned; a retry follows"); - } - } - } - Event::ControlRequested { .. } => { - if let Some(Derived::ControlRequested { - deliverable: true, - answer: Some(answer), - }) = &event.derived - { - let firing_key = event.subject.as_ref().and_then(|subject| { - subject - .firing - .map(|firing| stage_key(execution.raw(), firing.raw())) - }); - self.close_questions(answer.question.as_deref(), firing_key.as_deref(), at); - } - } - // ── Sandbox: the instance (VIEWS.md "Sandbox") ────────────────── - // The run's sandbox is the root invocation's scope. A child - // invocation's scope (a parallel branch) shares or owns another - // one and is not the run's; a re-acquisition (a resume, a - // replaced sandbox) names the current instance. - Event::ScopeAcquired { - sandbox, - duration_ms, - .. - } => { - if let Some(projection) = self.root_scope_projection(event) { - let plan = sandbox_plan_of(projection); - projection.sandbox = Some(RunSandbox::ready( - plan.clone(), - sandbox_instance(&plan, sandbox, *duration_ms), - )); - } - } - Event::ScopeFailed { - provider, - error, - causes, - duration_ms, - .. - } => { - if let Some(projection) = self.root_scope_projection(event) { - let plan = sandbox_plan_of(projection); - let provider = provider - .as_deref() - .and_then(provider_kind) - .unwrap_or_else(|| plan.provider.clone()); - projection.sandbox = Some(RunSandbox::failed(plan, RunSandboxFailure { - provider: provider.to_string(), - error: error.clone(), - causes: causes.clone(), - duration_ms: *duration_ms, - })); - } - } - Event::ExecutionStarted { .. } - | Event::TokenEmitted { .. } - | Event::RoutingResolved { .. } - | Event::RouteApplied { .. } - | Event::RetryElapsed { .. } - | Event::NodeExpanded { .. } - | Event::CancelRequested { .. } - | Event::KillRequested { .. } => {} - } - } - - /// The projection, when `event` is a scope record of the root - /// invocation: the run's own sandbox, not a child invocation's. - fn root_scope_projection(&mut self, event: &RunEvent) -> Option<&mut RunProjection> { - let root = self.state.root?; - if event.context.invocation.map(InvocationId::raw) != Some(root) { - return None; - } - self.projection.as_mut() - } - - fn fold_progress( - &mut self, - execution: ExecutionId, - event: &RunEvent, - ev: &StepEvent, - at: DateTime, - ) { - match ev { - StepEvent::Log { line, .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let output = stage.output.get_or_insert_default(); - output.push_str(line); - output.push('\n'); - stage.output_bytes = Some(output.len() as u64); - stage.live_streaming = Some(true); - } - } - StepEvent::Artifact { .. } => {} - StepEvent::Custom(payload) => { - if let Some(parsed) = event.parsed() { - self.fold_parsed(execution, event, parsed, at); - return; - } - let kind = payload.get("kind").and_then(Value::as_str).unwrap_or(""); - match kind { - "pebble" => self.fold_pebble(execution, event, payload, at), - "attractor.prompt" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.prompt = payload - .get("prompt") - .and_then(Value::as_str) - .map(str::to_string); - let model = payload.get("model").and_then(Value::as_str); - if let Some(model) = model { - let (provider, model_id) = split_model(model); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_PROMPT.to_string(), - provider: provider.map(str::to_string), - model: Some(model_id.to_string()), - reasoning_effort: None, - speed: None, - }); - stage.model = model_ref(provider, model_id); - } - } - } - "attractor.prompt.completed" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.response = payload - .get("response") - .and_then(Value::as_str) - .map(str::to_string); - if let Some(usage) = usage_of(payload.get("usage")) { - stage.usage = usage; - } - } - } - "attractor.fallback.plan" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let route = payload - .get("routes") - .and_then(Value::as_array) - .and_then(|routes| routes.first()); - if let Some(route) = route { - let provider = route.get("provider").and_then(Value::as_str); - let model = route.get("model").and_then(Value::as_str); - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.map(str::to_string), - model: model.map(str::to_string), - reasoning_effort: None, - speed: None, - }); - if let Some(model) = model { - stage.model = model_ref(provider, model); - } - } - } - } - // The tools a native session was offered, once per - // session (VIEWS.md "Agent activity", tools available): - // the stage's list is the union over its sessions, by - // name, in the order the sessions listed them. - "attractor.tools" => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - let tools = payload.get("tools").and_then(Value::as_array); - for tool in tools.into_iter().flatten() { - let Some(summary) = tool_summary(tool) else { - continue; - }; - if !stage - .agent_tools - .iter() - .any(|known| known.name == summary.name) - { - stage.agent_tools.push(summary); - } - } - } - } - "attractor.parallel.branch.started" => { - let invocation = payload.get("invocation").and_then(Value::as_u64); - let index = payload - .get("index") - .and_then(Value::as_u64) - .and_then(|index| u32::try_from(index).ok()); - let fork_firing = payload - .get("occurrence") - .and_then(|occurrence| occurrence.get("firing")) - .and_then(Value::as_u64); - if let (Some(invocation), Some(index), Some(fork_firing)) = - (invocation, index, fork_firing) - { - let group = self - .state - .stages - .get(&stage_key(execution.raw(), fork_firing)) - .map(|stage| stage.stage_id.clone()); - if let Some(group) = group { - self.state.invocations.entry(invocation).or_default().branch = - Some((group, index)); - } - } - } - _ => {} - } - } - } - } - - fn fold_parsed( - &mut self, - execution: ExecutionId, - event: &RunEvent, - parsed: &Parsed, - at: DateTime, - ) { - match parsed { - Parsed::Question { question } => { - let Some(subject) = event.subject.as_ref() else { - return; - }; - let Some(firing) = subject.firing else { - return; - }; - let key = stage_key(execution.raw(), firing.raw()); - let label = self.state.stages.get(&key).map_or_else( - || subject.node.name.to_string(), - |stage| stage.stage_id.to_string(), - ); - self.state.questions.insert(question.id.clone(), key); - let Some(projection) = self.projection.as_mut() else { - return; - }; - projection - .pending_interviews - .insert(question.id.clone(), PendingInterviewRecord { - question: InterviewQuestionRecord { - id: question.id.clone(), - text: question.text.clone(), - stage: label, - question_type: question_type(question), - options: question - .options - .iter() - .map(|option| InterviewOption { - key: option.key.clone(), - label: option.label.clone(), - description: option.description.clone(), - preview: option.preview.clone(), - }) - .collect(), - allow_freeform: question.freeform, - timeout_seconds: question - .timeout_ms - .map(|timeout| timeout as f64 / 1000.0), - context_display: question.context.clone(), - review_target: question.reference.as_ref().and_then(review_target), - }, - started_at: at, - }); - apply_status( - projection, - RunStatus::Blocked { - blocked_reason: BlockedReason::HumanInputRequired, - }, - at, - ); - } - Parsed::QuestionExpired { expired } => { - self.close_questions(Some(expired.question.as_str()), None, at); - } - Parsed::Note { .. } => {} - } - } - - /// Close one question by id, or every question of a firing, and unblock - /// the run when none is left. - fn close_questions( - &mut self, - question: Option<&str>, - firing_key: Option<&str>, - at: DateTime, - ) { - let closed: Vec = match (question, firing_key) { - (Some(question), _) => vec![question.to_string()], - (None, Some(key)) => self - .state - .questions - .iter() - .filter(|(_, asked_by)| asked_by.as_str() == key) - .map(|(id, _)| id.clone()) - .collect(), - (None, None) => Vec::new(), - }; - for id in &closed { - self.state.questions.remove(id); - } - let Some(projection) = self.projection.as_mut() else { - return; - }; - for id in &closed { - projection.pending_interviews.remove(id); - } - if projection.pending_interviews.is_empty() - && matches!(projection.status, RunStatus::Blocked { .. }) - { - apply_status(projection, RunStatus::Running, at); - } - } - - fn fold_pebble( - &mut self, - execution: ExecutionId, - event: &RunEvent, - payload: &Value, - at: DateTime, - ) { - let Some(envelope) = payload.get("event") else { - return; - }; - let envelope: CodingAgentEvent = match serde_json::from_value(envelope.clone()) { - Ok(envelope) => envelope, - Err(error) => { - debug!(error = %error, "a pebble envelope did not decode; skipped"); - return; - } - }; - let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { - return; - }; - let agent = stage.agent.get_or_insert_default(); - agent.apply(&envelope); - if stage.completion.is_none() { - stage.usage = agent.usage.saturating_add(agent.descendant_usage()); - } - // A tool the stage's list names was called, by any of its sessions. - if let CodingEvent::ToolCallStarted { tool_name, .. } = &envelope.event { - if let Some(tool) = stage - .agent_tools - .iter_mut() - .find(|tool| tool.name == *tool_name) - { - tool.invoked = true; - } - } - let is_root = envelope.parent_session_id.is_none(); - #[expect( - clippy::wildcard_enum_match_arm, - reason = "pebble's event vocabulary is non-exhaustive and only some events project" - )] - match &envelope.event { - CodingEvent::SessionStarted { - provider, model, .. - } if is_root => { - stage.provider_used = Some(StageModelUsage { - mode: StageModelUsage::MODE_AGENT.to_string(), - provider: provider.clone(), - model: model.clone(), - reasoning_effort: None, - speed: None, - }); - if let Some(model) = model.as_deref() { - stage.model = model_ref(provider.as_deref(), model); - } - } - CodingEvent::LlmRequestStarted { requested_model } if is_root => { - stage.inference = Some(StageInferenceProjection { - session_id: envelope.session_id.clone(), - started_at: at, - requested_model: requested_model.clone(), - first_output_at: None, - first_output_kind: None, - retries: 0, - }); - } - CodingEvent::LlmFirstOutput { kind } => { - if let Some(inference) = stage.inference.as_mut() { - if inference.session_id == envelope.session_id { - inference.first_output_at = Some(at); - inference.first_output_kind = Some(*kind); - } - } - } - CodingEvent::LlmRetry { .. } => { - if let Some(inference) = stage.inference.as_mut() { - if inference.session_id == envelope.session_id { - inference.retries = inference.retries.saturating_add(1); - inference.first_output_at = None; - inference.first_output_kind = None; - } - } - } - CodingEvent::AssistantMessage { model, .. } => { - if is_root { - if let Some(provider) = stage - .provider_used - .as_ref() - .and_then(|used| used.provider.as_deref()) - { - stage.model = model_ref(Some(provider), model); - } - } - close_inference(stage, &envelope.session_id, at); - } - CodingEvent::Error { .. } | CodingEvent::RoundInterrupted { .. } => { - close_inference(stage, &envelope.session_id, at); - } - CodingEvent::SessionEnded => { - close_inference(stage, &envelope.session_id, at); - stage.close_tool_batch_for_session(&envelope.session_id, at); - } - CodingEvent::ToolCallStarted { tool_call_id, .. } if is_root => { - stage.open_tool_call(envelope.session_id.clone(), tool_call_id.clone(), at); - } - CodingEvent::ToolCallCompleted { tool_call_id, .. } if is_root => { - stage.close_tool_call(&envelope.session_id, tool_call_id, at); - } - _ => {} - } - } - - fn fold_view(&mut self, view: &ViewEvent, event: &RunEvent, at: DateTime) { - let Some(execution) = event.context.execution else { - return; - }; - match view { - ViewEvent::VisitStarted { .. } => { - let Some(subject) = event.subject.as_ref() else { - return; - }; - self.start_visit(execution, subject, at); - } - ViewEvent::WaitStateChanged { state } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - match state { - WaitState::AwaitingAdmission => { - if stage.state == StageState::Running { - stage.state = StageState::Pending; - } - } - WaitState::Running | WaitState::AwaitingAnswer | WaitState::Cancelling => { - stage.state = StageState::Running; - } - WaitState::AwaitingRetry => stage.state = StageState::Retrying, - } - } - } - ViewEvent::RetryScheduled { .. } => { - if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { - stage.state = StageState::Retrying; - } - } - ViewEvent::VisitCompleted { - outcome, - executed, - attempts, - } => { - let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { - return; - }; - stage.state = match outcome.status { - Status::Success => StageState::Succeeded, - Status::PartialSuccess { .. } => StageState::PartiallySucceeded, - Status::Failure(_) | Status::TimedOut => StageState::Failed, - Status::Skipped => StageState::Skipped, - Status::Cancelled => StageState::Cancelled, - }; - if stage.completion.is_none() || !*executed { - stage.completion = Some(StageCompletion { - outcome: stage_outcome(&outcome.status), - notes: None, - failure_reason: failure_message(&outcome.status), - timestamp: at, - }); - } - if stage.timing.is_none() { - let wall = stage - .started_at - .map_or(0, |started| timing::elapsed_ms(started, at)); - stage.set_authoritative_timing(StageTiming::new(wall, 0, 0)); - } - debug!(attempts, "visit completed"); - } - ViewEvent::ForkCompleted { - occurrence, - results, - .. - } => { - let key = stage_key(occurrence.execution.raw(), occurrence.firing.raw()); - let Some(stage_id) = self - .state - .stages - .get(&key) - .map(|stage| stage.stage_id.clone()) - else { - return; - }; - let Some(projection) = self.projection.as_mut() else { - return; - }; - if let Some(stage) = projection.stage_mut(&stage_id) { - stage.parallel_results = Some( - results - .iter() - .map(|result| ParallelBranchResult { - id: result.node.name.to_string(), - index: Some(result.branch.index as usize), - item_label: None, - status: stage_outcome(&result.status), - context_updates: BTreeMap::new(), - }) - .collect(), - ); - } - } - ViewEvent::ForkStarted { .. } - | ViewEvent::BranchCompleted { .. } - | ViewEvent::RunStalled { .. } => {} - } - } - - /// A firing exists: register its stage and, when it is a logical stage, - /// show it. - fn start_visit(&mut self, execution: ExecutionId, subject: &Subject, at: DateTime) { - let Some(firing) = subject.firing else { - return; - }; - let key = stage_key(execution.raw(), firing.raw()); - if self.state.stages.contains_key(&key) { - return; - } - let node_name = subject.node.name.to_string(); - let visit = visit_of(subject); - let meta_kind = node_meta_kind(&subject.node); - let shown = is_shown(&subject.node); - // Only a shown stage takes a label: a lowering node (a branch's - // parent-side delegate shares its target's name) never competes with - // the stage it stands for. - let mut stage_id = StageId::new(node_name.clone(), visit); - if shown { - stage_id = stage_label(&node_name, visit, execution, &self.state.labels); - self.state.labels.insert(stage_id.to_string()); - } - self.state.stages.insert(key, StageRef { - stage_id: stage_id.clone(), - shown, - node_name, - visit, - }); - if !shown { - return; - } - let branch = self - .state - .executions - .get(&execution.raw()) - .and_then(|invocation| self.state.invocations.get(invocation)) - .and_then(|invocation| invocation.branch.clone()); - let Some(projection) = self.projection.as_mut() else { - return; - }; - let since_created = at - .signed_duration_since(projection.spec.run_id.created_at()) - .num_milliseconds() - .max(0); - let ordinal = u32::try_from(since_created) - .unwrap_or(u32::MAX - 1) - .saturating_add(1); - let stage = projection.stage_entry(stage_id.node_id(), visit, first_event_seq(ordinal)); - stage.handler = Some(StageHandler::from_handler_type(Some(meta_kind))); - stage.started_at = Some(at); - stage.graph_visit = Some(visit); - stage.state = StageState::Pending; - stage.parallel_branch_id = branch.map(|(group, index)| ParallelBranchId::new(group, index)); - } - - /// The shown stage an event's subject firing belongs to. - fn stage_of( - &mut self, - execution: ExecutionId, - subject: Option<&Subject>, - ) -> Option<&mut StageProjection> { - let firing = subject?.firing?; - let stage = self - .state - .stages - .get(&stage_key(execution.raw(), firing.raw()))?; - if !stage.shown { - return None; - } - let stage_id = stage.stage_id.clone(); - self.projection.as_mut()?.stage_mut(&stage_id) - } -} - -impl Default for RunView { - fn default() -> Self { - Self::new() - } -} - -// ── Lifecycle ─────────────────────────────────────────────────────────── - -fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, at: DateTime) { - use RunLifecycleKind as Kind; - match record.transition { - Kind::Submitted => apply_status(projection, RunStatus::Submitted, at), - Kind::StartRequested => {} - Kind::Unpaused => { - let status = match projection.status { - RunStatus::Paused { - prior_block: Some(blocked_reason), - } => RunStatus::Blocked { blocked_reason }, - _ => RunStatus::Running, - }; - apply_status(projection, status, at); - if projection.pending_control == Some(RunControlAction::Unpause) { - projection.pending_control = None; - } - } - Kind::Pending => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - projection.approval = Some(RunApproval { - state: RunApprovalState::Pending, - requested_at: at, - decided_at: None, - denial_reason: None, - }); - } - Kind::Approved => { - if let Some(approval) = projection.approval.as_mut() { - approval.state = RunApprovalState::Approved; - approval.decided_at = Some(at); - } - } - Kind::Denied => { - if let Some(approval) = projection.approval.as_mut() { - approval.state = RunApprovalState::Denied; - approval.decided_at = Some(at); - approval.denial_reason.clone_from(&record.reason); - } - apply_status( - projection, - RunStatus::Failed { - reason: FailureReason::ApprovalDenied, - }, - at, - ); - } - Kind::Runnable => { - // A run left in flight by a restart goes back to the queue: the - // resume's `runnable` steps back from wherever the run stood. - let in_flight = matches!( - projection.status, - RunStatus::Starting - | RunStatus::Running - | RunStatus::Blocked { .. } - | RunStatus::Paused { .. } - ); - if in_flight && record.status == Some(RunStatus::Runnable) { - projection.status = RunStatus::Runnable; - projection.status_updated_at = at; - } else if let Some(status) = record.status { - apply_status(projection, status, at); - } - } - Kind::Blocked => { - // A block that lands while the run is paused waits behind the - // pause: the unpause restores it. - match (projection.status, record.status) { - (RunStatus::Paused { .. }, Some(RunStatus::Blocked { blocked_reason })) => { - apply_status( - projection, - RunStatus::Paused { - prior_block: Some(blocked_reason), - }, - at, - ); - } - (_, Some(status)) => apply_status(projection, status, at), - (_, None) => {} - } - } - Kind::Unblocked => { - let status = match projection.status { - RunStatus::Paused { .. } => RunStatus::Paused { prior_block: None }, - _ => record.status.unwrap_or(RunStatus::Running), - }; - apply_status(projection, status, at); - } - Kind::Starting | Kind::Running | Kind::Removing | Kind::Dead => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - } - Kind::Paused => { - let prior_block = match projection.status { - RunStatus::Blocked { blocked_reason } => Some(blocked_reason), - RunStatus::Paused { prior_block } => prior_block, - _ => None, - }; - apply_status(projection, RunStatus::Paused { prior_block }, at); - if projection.pending_control == Some(RunControlAction::Pause) { - projection.pending_control = None; - } - } - Kind::Succeeded | Kind::Failed => { - if let Some(status) = record.status { - apply_status(projection, status, at); - } - projection.pending_control = None; - if projection.conclusion.is_none() { - let (outcome, failure) = match record.status { - Some(RunStatus::Failed { reason }) => ( - StageOutcome::Failed { - retry_requested: false, - }, - Some(RunFailure { - reason, - detail: FailureDetail::new( - record - .reason - .clone() - .unwrap_or_else(|| "the run failed".to_string()), - FailureCategory::Deterministic, - ), - }), - ), - _ => (StageOutcome::Succeeded, None), - }; - projection.conclusion = Some(Conclusion { - timestamp: at, - status: outcome, - timing: RunTiming::default(), - failure, - final_git_commit_sha: None, - stages: Vec::new(), - usage: None, - total_retries: 0, - diff: RunDiff::default(), - }); - } - } - Kind::CancelRequested => projection.pending_control = Some(RunControlAction::Cancel), - Kind::PauseRequested => projection.pending_control = Some(RunControlAction::Pause), - Kind::UnpauseRequested => projection.pending_control = Some(RunControlAction::Unpause), - } -} - -/// Apply a status transition; one the lifecycle refuses is logged and -/// skipped, since the view never fails the run. -fn apply_status(projection: &mut RunProjection, status: RunStatus, at: DateTime) { - if let Err(error) = projection.try_apply_status(status, at) { - debug!(error = %error, "status transition not applied to the Petri projection"); - } -} - -fn touch(projection: &mut RunProjection, at: DateTime) { - if at > projection.last_event_at { - projection.last_event_at = at; - } -} - -// ── Helpers ───────────────────────────────────────────────────────────── - -/// The key of a stage: its execution and firing. -#[must_use] -pub fn stage_key(execution: u64, firing: u64) -> String { - format!("{execution}:{firing}") -} - -/// Which firing of its node a subject is, 1-based. -#[must_use] -pub fn visit_of(subject: &Subject) -> u32 { - subject.visit.unwrap_or(1).max(1) -} - -/// The role a frontend gave a node under `meta.kind`, or the empty string. -fn node_meta_kind(node: &NodeRef) -> &str { - node.meta.get("kind").and_then(Value::as_str).unwrap_or("") -} - -/// Whether a node is a logical stage the projection shows, or a lowering -/// node it keeps off the list: one a frontend marked synthetic, or a -/// parallel branch's delegate. -#[must_use] -pub fn is_shown(node: &NodeRef) -> bool { - let synthetic = node - .meta - .get("synthetic") - .and_then(Value::as_bool) - .unwrap_or(false); - !synthetic && node_meta_kind(node) != "parallel.branch" -} - -/// The label a shown firing takes, which is the stage id the projection -/// keys it by: `node@visit`, or `node/e@visit` when another -/// execution's firing already took that label. `taken` is every label given -/// so far; the caller adds the one returned. The interview adapter labels a -/// question's stage through this same rule, so the stage a question names -/// is the stage the projection shows. -#[must_use] -pub fn stage_label( - node_name: &str, - visit: u32, - execution: ExecutionId, - taken: &BTreeSet, -) -> StageId { - let stage_id = StageId::new(node_name.to_string(), visit); - if taken.contains(&stage_id.to_string()) { - return StageId::new(format!("{node_name}/e{}", execution.raw()), visit); - } - stage_id -} - -/// The fork firing and branch index a branch child's call slot names: -/// `branch:@::`. -fn branch_slot(slot: &str) -> Option<(u64, u32)> { - let rest = slot.strip_prefix("branch:")?; - let mut parts = rest.splitn(3, ':'); - let fork = parts.next()?; - let index = parts.next()?.parse::().ok()?; - let firing = fork.rsplit_once('@')?.1.parse::().ok()?; - Some((firing, index)) -} - -fn millis(recorded_at: u64) -> DateTime { - Utc.timestamp_millis_opt(i64::try_from(recorded_at).unwrap_or(i64::MAX)) - .single() - .unwrap_or_default() -} - -fn sandbox_plan(settings: &RunEnvironmentSettings) -> RunSandboxPlan { - RunSandboxPlan { - provider: settings.provider.clone(), - image: (settings.provider == SandboxProviderKind::DOCKER) - .then(|| settings.image.docker.clone()) - .flatten() - .filter(|image| !image.is_empty()), - snapshot: None, - } -} - -/// The plan the projection's sandbox carries, or the one its environment -/// settings give when no sandbox was projected yet. -fn sandbox_plan_of(projection: &RunProjection) -> RunSandboxPlan { - projection.sandbox.as_ref().map_or_else( - || sandbox_plan(&projection.spec.settings.run.environment), - |sandbox| sandbox.plan().clone(), - ) -} - -/// Fabro's name for the provider Petri's `scope.acquired` names: Petri's -/// `host` is Fabro's `local`; every other kind is spelled the same. `None` -/// for a name that is no provider kind. -fn provider_kind(provider: &str) -> Option { - if provider == "host" { - return Some(SandboxProviderKind::LOCAL); - } - SandboxProviderKind::try_new(provider).ok() -} - -/// The run's sandbox instance from Petri's record of the scope's -/// acquisition: the provider, the provider's id for the sandbox (what a -/// reconnect attaches by), its image and snapshot when the provider knows -/// them, the working directory, and how long the acquisition took. The -/// clone fields stay unset: Petri's checkout copies the bound repository -/// into the workspace and is not a clone Fabro made, and the workspace -/// roots are the provider's own layout, read live. `retained` waits for -/// the scope's release. -fn sandbox_instance( - plan: &RunSandboxPlan, - sandbox: &SandboxInstance, - ready_duration_ms: u64, -) -> RunSandboxInstance { - RunSandboxInstance { - provider: provider_kind(&sandbox.provider) - .unwrap_or_else(|| plan.provider.clone()), - image: sandbox - .image - .as_ref() - .map(ToString::to_string) - .or_else(|| plan.image.clone()), - snapshot: sandbox.snapshot.as_ref().map(ToString::to_string), - runtime: RunSandboxRuntime { - id: sandbox.instance.to_string(), - working_directory: sandbox.working_directory.to_string(), - repo_cloned: None, - clone_origin_url: None, - clone_branch: None, - workspace_root: None, - repos_root: None, - primary_repo_path: None, - primary_repo_link: None, - }, - ready_duration_ms: Some(ready_duration_ms), - retained: None, - } -} - -fn stage_outcome(status: &Status) -> StageOutcome { - match status { - Status::Success => StageOutcome::Succeeded, - Status::PartialSuccess { .. } => StageOutcome::PartiallySucceeded, - Status::Failure(info) => StageOutcome::Failed { - retry_requested: info.class.as_str() == "retry_requested", - }, - Status::Skipped => StageOutcome::Skipped, - Status::Cancelled | Status::TimedOut => StageOutcome::Failed { - retry_requested: false, - }, - } -} - -fn failure_message(status: &Status) -> Option { - match status { - Status::Failure(info) - | Status::PartialSuccess { - underlying: Some(info), - } => Some(info.message.clone()), - Status::TimedOut => Some("the step timed out".to_string()), - Status::Cancelled => Some("the step was cancelled".to_string()), - Status::Success | Status::PartialSuccess { underlying: None } | Status::Skipped => None, - } -} - -/// The finished attempt's metrics onto its stage: the timing and the usage -/// the backend reported. -/// One tool of an `attractor.tools` payload as the stage's list carries -/// it: the name and description as recorded, Pebble's `source` as it is, -/// and Pebble's behavioural category where Petri's says which (a -/// sub-agent tool); every other tool is `other`, because the payload -/// carries Petri's origin category (`builtin`, `mcp`, `host`, `question`), -/// not Pebble's permission class. `invoked` starts false and flips on the -/// session's `ToolCallStarted`. -fn tool_summary(tool: &Value) -> Option { - let name = tool.get("name").and_then(Value::as_str)?; - let description = tool - .get("description") - .and_then(Value::as_str) - .unwrap_or_default(); - let source = tool - .get("source") - .cloned() - .and_then(|source| serde_json::from_value::(source).ok()) - .unwrap_or_default(); - let category = match tool.get("category").and_then(Value::as_str) { - Some("subagent") => ToolCategory::Subagent, - _ => ToolCategory::Other, - }; - Some(ToolSummary { - name: name.to_string(), - description: description.to_string(), - source, - category, - invoked: false, - }) -} - -/// The question's `reference` as Fabro's review target, when it is one -/// Fabro's validation admits (a `document`, or a reference without a kind, -/// with a label and an absolute HTTP URL within Fabro's limits). -fn review_target(reference: &QuestionReference) -> Option { - let kind = match reference.kind.as_deref() { - Some("document") | None => ReviewTargetKind::Document, - Some(_) => return None, - }; - ReviewTarget::new(reference.label.clone(), reference.url.clone(), kind).ok() -} - -fn apply_metrics(stage: &mut StageProjection, metrics: &Metrics) { - let custom = &metrics.custom; - let inference = custom - .get("pebble.inference_ms") - .and_then(Value::as_u64) - .unwrap_or(0); - let tool = custom - .get("pebble.tool_ms") - .and_then(Value::as_u64) - .unwrap_or(0); - let wall = metrics.duration_ms.unwrap_or(0); - let (inference, tool) = match stage.handler { - Some(StageHandler::Prompt) => (wall, 0), - Some(StageHandler::Command) => (0, wall), - _ => (inference, tool), - }; - stage.set_authoritative_timing(StageTiming::new(wall, inference, tool).clamped_to_wall()); - if let Some(usage) = - usage_of(custom.get("pebble.usage")).or_else(|| usage_of(custom.get("prompt.usage"))) - { - stage.usage = usage; - } - if let Some(sessions) = custom - .get("pebble.subagents") - .and_then(|subagents| subagents.get("sessions")) - .and_then(Value::as_array) - { - let mut by_model: Vec = Vec::new(); - for session in sessions { - let provider = session.get("provider").and_then(Value::as_str); - let model = session.get("model").and_then(Value::as_str); - let Some(usage) = usage_of(session.get("usage")) else { - continue; - }; - let Some(model) = model.and_then(|model| model_ref(provider, model)) else { - continue; - }; - if let Some(entry) = by_model.iter_mut().find(|entry| entry.model == model) { - entry.usage = entry.usage.saturating_add(usage); - } else { - by_model.push(ModelUsage::new(model, usage)); - } - } - if !by_model.is_empty() { - stage.usage_by_model = by_model; - } - } -} - -fn usage_of(value: Option<&Value>) -> Option { - serde_json::from_value(value?.clone()).ok() -} - -/// `provider/model` into its parts, or the model alone. -fn split_model(model: &str) -> (Option<&str>, &str) { - match model.split_once('/') { - Some((provider, model)) if !provider.is_empty() && !model.is_empty() => { - (Some(provider), model) - } - _ => (None, model), - } -} - -fn model_ref(provider: Option<&str>, model: &str) -> Option { - let provider = provider.filter(|provider| !provider.is_empty())?; - Some(ModelRef::new( - ProviderId::new(provider), - ModelId::new(model), - )) -} - -fn close_inference(stage: &mut StageProjection, session_id: &str, at: DateTime) { - let open = stage - .inference - .as_ref() - .is_some_and(|inference| inference.session_id == session_id); - if !open { - return; - } - if let Some(inference) = stage.inference.take() { - stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, at)); - } -} - -/// The run id a Petri run key names. -#[must_use] -pub fn run_id_of(key: &str) -> Option { - key.parse().ok() -} - -#[cfg(test)] -mod tests { - use fabro_store::platform_records::RunCreatedRecord; - use fabro_types::test_support as types_support; - use petri_runtime::driver::BranchRole; - use petri_runtime::ir::{FiringId, NodeId}; - - use super::*; - - #[test] - fn a_branch_slot_names_the_fork_firing_and_the_index() { - assert_eq!(branch_slot("branch:fan@7:2:review"), Some((7, 2))); - assert_eq!(branch_slot("branch:fan@7:x:review"), None); - assert_eq!(branch_slot("child:0"), None); - } - - #[test] - fn a_model_selector_splits_into_provider_and_model() { - assert_eq!(split_model("openai/gpt-5.4"), (Some("openai"), "gpt-5.4")); - assert_eq!(split_model("gpt-5.4"), (None, "gpt-5.4")); - assert!(model_ref(None, "gpt-5.4").is_none()); - assert!(model_ref(Some("openai"), "gpt-5.4").is_some()); - } - - #[test] - fn a_taken_label_is_made_unique_by_the_execution() { - let mut view = RunView::new(); - let created = StoredPlatformRecord { - seq: 1, - recorded_at: 1_000, - record: PlatformRecord::RunCreated(RunCreatedRecord { - spec: types_support::test_run_spec(), - title: Some("A run".to_string()), - parent_id: None, - retried_from: None, - web_url: None, - }), - position: None, - }; - view.fold(&Item::Platform(&created), 1); - let subject = |name: &str| Subject { - node: NodeRef { - id: NodeId::new(1), - name: name.into(), - kind: "attractor/command".into(), - meta: serde_json::json!({ "kind": "command" }), - }, - firing: Some(FiringId::new(4)), - visit: Some(1), - attempt: None, - generation: None, - branch: BranchRole::None, - }; - view.start_visit(ExecutionId::new(1), &subject("build"), millis(2_000)); - view.start_visit(ExecutionId::new(2), &subject("build"), millis(3_000)); - let labels: Vec = view - .projection() - .expect("the run was created") - .iter_stages() - .map(|(id, _)| id.to_string()) - .collect(); - assert_eq!(labels, vec!["build@1", "build/e2@1"]); - assert_eq!(view.state.stages.len(), 2); - } -} diff --git a/lib/components/fabro-petri/src/projection/coordinator.rs b/lib/components/fabro-petri/src/projection/coordinator.rs new file mode 100644 index 000000000..eb51c0529 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/coordinator.rs @@ -0,0 +1,241 @@ +//! Petri's coordinator events folded into the view: the run's start, its +//! invocations and executions, pause and unpause, its finish, and the +//! release of its sandbox (VIEWS.md "Run", "Sandbox"). + +use chrono::{DateTime, Utc}; +use fabro_types::{ + Conclusion, FailureCategory, FailureDetail, FailureReason, RunControlAction, RunFailure, + RunSandbox, RunStatus, RunTiming, StageOutcome, StartRecord, SuccessReason, usage_rollup, +}; +use petri_execution::CoordinatorEvent; +use petri_execution::events::RunEvent; + +use super::{FiringKey, InvocationRef, RunView, apply_status, settle_control}; + +impl RunView { + pub(super) fn fold_coordinator( + &mut self, + record: &CoordinatorEvent, + event: &RunEvent, + at: DateTime, + ) { + match record { + CoordinatorEvent::RunStarted { + root, forked_from, .. + } => { + self.state.root = Some(root.raw()); + self.state.started_at = Some(event.recorded_at); + if let Some(projection) = self.projection.as_mut() { + // A fork's declaration names its source; a parse failure + // means the source was not a Fabro run, which the + // projection cannot show. + projection.forked_from = forked_from.as_ref().and_then(|origin| { + Some(fabro_types::ForkOrigin { + source_run_id: origin.source.as_str().parse().ok()?, + execution: origin.position.execution.raw(), + firing: origin.position.firing.raw(), + rerun_last: origin.rerun_last, + }) + }); + apply_status(projection, RunStatus::Running, at); + projection.start = Some(StartRecord { + start_time: at, + run_branch: self.state.run_branch.clone(), + base_sha: self.state.base_sha.clone(), + }); + // The scope's sandbox is acquired next; `scope.acquired` + // or `scope.failed` settles it. + if let Some(sandbox) = projection.sandbox.take() { + projection.sandbox = Some(RunSandbox::initializing(sandbox.plan().clone())); + } + } + } + CoordinatorEvent::InvocationDeclared { invocation, .. } => { + let mut info = InvocationRef::default(); + if let Some(parent) = &event.context.parent { + info.parent = Some((parent.execution.raw(), parent.firing.raw())); + if let Some((fork_firing, index)) = branch_slot(&parent.slot) { + let group = self + .state + .stages + .get(&FiringKey::new(parent.execution.raw(), fork_firing)) + .map(|stage| stage.stage_id.clone()); + if let Some(group) = group { + info.branch = Some((group, index)); + } + } + } + self.state.invocations.insert(invocation.raw(), info); + } + CoordinatorEvent::ExecutionDeclared { + execution, + invocation, + .. + } => { + self.state + .executions + .insert(execution.raw(), invocation.raw()); + } + CoordinatorEvent::InvocationFinished { invocation, result } => { + let info = self.state.invocations.entry(invocation.raw()).or_default(); + info.failure = result + .failure + .as_ref() + .map(|failure| failure.message.clone()); + info.output = Some(result.output.clone()); + } + CoordinatorEvent::RunPaused => { + if let Some(projection) = self.projection.as_mut() { + apply_status(projection, projection.status.paused(), at); + settle_control(projection, RunControlAction::Pause); + } + } + CoordinatorEvent::RunUnpaused => { + if let Some(projection) = self.projection.as_mut() { + apply_status(projection, projection.status.unpaused(), at); + settle_control(projection, RunControlAction::Unpause); + } + } + CoordinatorEvent::RunFinished { status } => { + self.state.finished = Some(status.to_string()); + self.conclude(status.to_string().as_str(), at); + } + // ── Sandbox: the retention outcome (VIEWS.md "Sandbox") ───────── + // The instance stays on `Run.sandbox`: it names what ran, and + // `retained` says whether it still exists. + CoordinatorEvent::ScopeReleased { + invocation, + retained, + .. + } => { + if Some(invocation.raw()) == self.state.root { + self.state.sandbox_retained = Some(*retained); + if let Some(sandbox) = self + .projection + .as_mut() + .and_then(|projection| projection.sandbox.as_mut()) + { + sandbox.set_retained(*retained); + } + } + } + CoordinatorEvent::GraphRegistered { .. } + | CoordinatorEvent::ExecutionFinished { .. } + | CoordinatorEvent::InvocationCancelRequested { .. } + | CoordinatorEvent::RunNoteRecorded { .. } => {} + } + } + + /// The run's conclusion, from its recorded finish and what the stages + /// The run's conclusion, from its recorded finish and what the stages + /// summed to. + fn conclude(&mut self, status: &str, at: DateTime) { + let Some(projection) = self.projection.as_mut() else { + return; + }; + let root = self + .state + .root + .and_then(|root| self.state.invocations.get(&root)); + let failure_message = root.and_then(|root| root.failure.clone()); + let (run_status, outcome, failure) = match status { + "success" => ( + RunStatus::Succeeded { + reason: SuccessReason::Completed, + }, + StageOutcome::Succeeded, + None, + ), + "cancelled" => ( + RunStatus::Failed { + reason: FailureReason::Cancelled, + }, + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason: FailureReason::Cancelled, + detail: FailureDetail::new( + failure_message + .clone() + .unwrap_or_else(|| "the run was cancelled".to_string()), + FailureCategory::Canceled, + ), + }), + ), + _ => ( + RunStatus::Failed { + reason: FailureReason::WorkflowError, + }, + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason: FailureReason::WorkflowError, + detail: FailureDetail::new( + failure_message + .clone() + .unwrap_or_else(|| "the run failed".to_string()), + FailureCategory::Deterministic, + ), + }), + ), + }; + apply_status(projection, run_status, at); + projection.pending_control = None; + projection.pending_interviews.clear(); + let rollup = usage_rollup::usage_rollup_from_projection(projection); + let (stages, total_retries) = rollup.conclusion_stages(projection); + let wall_time_ms = self.state.started_at.map_or(0, |started| { + u64::try_from(at.timestamp_millis()) + .unwrap_or(0) + .saturating_sub(started) + }); + let timing = RunTiming::new( + wall_time_ms, + rollup.timing.inference_time_ms, + rollup.timing.tool_time_ms, + ); + let last_checkpoint = projection.checkpoints.last(); + projection.conclusion = Some(Conclusion { + timestamp: at, + status: outcome, + timing, + failure, + final_git_commit_sha: last_checkpoint + .and_then(|checkpoint| checkpoint.checkpoint.git_commit_sha.clone()), + stages, + usage: rollup.usage_if_present(), + total_retries, + diff: self + .state + .run_diff + .clone() + .or_else(|| last_checkpoint.map(|checkpoint| checkpoint.diff.clone())) + .unwrap_or_default(), + }); + } +} + +/// The fork firing and branch index a branch child's call slot names: +/// `branch:@::`. +fn branch_slot(slot: &str) -> Option<(u64, u32)> { + let rest = slot.strip_prefix("branch:")?; + let mut parts = rest.splitn(3, ':'); + let fork = parts.next()?; + let index = parts.next()?.parse::().ok()?; + let firing = fork.rsplit_once('@')?.1.parse::().ok()?; + Some((firing, index)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_branch_slot_names_the_fork_firing_and_the_index() { + assert_eq!(branch_slot("branch:fan@7:2:review"), Some((7, 2))); + assert_eq!(branch_slot("branch:fan@7:x:review"), None); + assert_eq!(branch_slot("child:0"), None); + } +} diff --git a/lib/components/fabro-petri/src/projection/engine.rs b/lib/components/fabro-petri/src/projection/engine.rs new file mode 100644 index 000000000..12fae7891 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/engine.rs @@ -0,0 +1,453 @@ +//! Petri's engine and view events folded into the stages: a firing's +//! visit, its attempts and their outcomes, its wait states, and the scope +//! its sandbox was acquired in (VIEWS.md "Stages", "Sandbox"). + +use std::collections::BTreeMap; + +use chrono::{DateTime, Utc}; +use fabro_types::{ + ModelUsage, ParallelBranchId, ParallelBranchResult, RunProjection, RunSandbox, + RunSandboxFailure, StageCompletion, StageHandler, StageId, StageOutcome, StageProjection, + StageState, StageTiming, first_event_seq, parse_blob_ref, timing, +}; +use petri_execution::events::{Derived, RunEvent, Subject, ViewEvent, WaitState}; +use petri_execution::{ExecutionId, InvocationId}; +use petri_runtime::engine::{Admission, Event}; +use petri_runtime::ir::{Metrics, Status}; +use serde_json::Value; +use tracing::debug; + +use super::model::{model_ref, usage_of}; +use super::sandbox::{provider_kind, sandbox_instance, sandbox_plan_of}; +use super::{FiringKey, RunView, StageRef, is_shown, node_meta_kind, stage_label, visit_of}; + +impl RunView { + pub(super) fn fold_engine(&mut self, engine: &Event, event: &RunEvent, at: DateTime) { + let Some(execution) = event.context.execution else { + return; + }; + match engine { + Event::AdmissionDecided { decision, .. } => { + if let Admission::Skip { outcome } = decision { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.state = StageState::Skipped; + stage.completion = Some(StageCompletion { + outcome: StageOutcome::Skipped, + ..completion(&outcome.status, at) + }); + } + } + } + Event::StepStarted { attempt, .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + if attempt.raw() > 1 { + stage.clear_live_timing(); + stage.output = None; + stage.output_bytes = None; + } + stage.state = StageState::Running; + stage.live_streaming = Some(true); + } + } + Event::StepProgressRecorded { ev, .. } => { + self.fold_progress(execution, event, ev, at); + } + Event::StepFinished { + firing, + attempt, + outcome, + } => { + self.state + .finished_firings + .insert(FiringKey::new(execution.raw(), firing.raw())); + let is_final = matches!( + event.derived, + Some(Derived::StepFinished { is_final: true, .. }) + ); + let node_name = event + .subject + .as_ref() + .map(|subject| subject.node.name.to_string()); + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + // The step's output: a string, or a command's `stdout`, + // either of which is a `blob://` reference when the + // step offloaded it. The reference stays as it is; the + // bytes it names are the live log's. + let output = outcome + .output + .as_str() + .or_else(|| outcome.output.get("stdout").and_then(Value::as_str)); + if let Some(output) = output { + if parse_blob_ref(output).is_none() { + stage.output_bytes = Some(output.len() as u64); + } + stage.output = Some(output.to_string()); + } + // A simulated step (a dry run) answers with its text. + let simulated = outcome + .output + .get("simulated") + .and_then(Value::as_bool) + .unwrap_or(false); + if simulated + && matches!( + stage.handler, + Some(StageHandler::Prompt | StageHandler::Agent) + ) + { + if let Some(text) = outcome.output.get("text").and_then(Value::as_str) { + stage.response = Some(text.to_string()); + } + } + // An agent's answer: the `response.` the step wrote + // into the run context, as the prompt step writes it. + if stage.handler == Some(StageHandler::Agent) { + let response = node_name + .as_deref() + .and_then(|name| { + outcome + .context_updates + .get(format!("response.{name}").as_str()) + }) + .and_then(Value::as_str) + .or_else(|| outcome.output.as_str()); + if let Some(response) = response { + stage.response = Some(response.to_string()); + } + } + stage.live_streaming = Some(false); + apply_metrics(stage, &outcome.metrics); + if is_final { + stage.completion = Some(completion(&outcome.status, at)); + stage.termination = Some(match outcome.status { + Status::TimedOut => fabro_types::CommandTermination::TimedOut, + Status::Cancelled => fabro_types::CommandTermination::Cancelled, + Status::Success + | Status::PartialSuccess { .. } + | Status::Failure(_) + | Status::Skipped => fabro_types::CommandTermination::Exited, + }); + } else { + stage.state = StageState::Retrying; + debug!(attempt = attempt.raw(), "attempt returned; a retry follows"); + } + } + } + Event::ControlRequested { .. } => { + if let Some(Derived::ControlRequested { + deliverable: true, + answer: Some(answer), + }) = &event.derived + { + self.close_questions( + answer.question.as_deref(), + FiringKey::of_event(event), + at, + ); + } + } + // ── Sandbox: the instance (VIEWS.md "Sandbox") ────────────────── + // The run's sandbox is the root invocation's scope. A child + // invocation's scope (a parallel branch) shares or owns another + // one and is not the run's; a re-acquisition (a resume, a + // replaced sandbox) names the current instance. + Event::ScopeAcquired { + sandbox, + duration_ms, + .. + } => { + if let Some(projection) = self.root_scope_projection(event) { + let plan = sandbox_plan_of(projection); + projection.sandbox = Some(RunSandbox::ready( + plan.clone(), + sandbox_instance(&plan, sandbox, *duration_ms), + )); + } + } + Event::ScopeFailed { + provider, + error, + causes, + duration_ms, + .. + } => { + if let Some(projection) = self.root_scope_projection(event) { + let plan = sandbox_plan_of(projection); + let provider = provider + .as_deref() + .and_then(provider_kind) + .unwrap_or_else(|| plan.provider.clone()); + projection.sandbox = Some(RunSandbox::failed(plan, RunSandboxFailure { + provider: provider.to_string(), + error: error.clone(), + causes: causes.clone(), + duration_ms: *duration_ms, + })); + } + } + Event::ExecutionStarted { .. } + | Event::TokenEmitted { .. } + | Event::RoutingResolved { .. } + | Event::RouteApplied { .. } + | Event::RetryElapsed { .. } + | Event::NodeExpanded { .. } + | Event::CancelRequested { .. } + | Event::KillRequested { .. } => {} + } + } + + /// The projection, when `event` is a scope record of the root + /// The projection, when `event` is a scope record of the root + /// invocation: the run's own sandbox, not a child invocation's. + fn root_scope_projection(&mut self, event: &RunEvent) -> Option<&mut RunProjection> { + let root = self.state.root?; + if event.context.invocation.map(InvocationId::raw) != Some(root) { + return None; + } + self.projection.as_mut() + } + + pub(super) fn fold_view(&mut self, view: &ViewEvent, event: &RunEvent, at: DateTime) { + let Some(execution) = event.context.execution else { + return; + }; + match view { + ViewEvent::VisitStarted { .. } => { + let Some(subject) = event.subject.as_ref() else { + return; + }; + self.start_visit(execution, subject, at); + } + ViewEvent::WaitStateChanged { state } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + match state { + WaitState::AwaitingAdmission => { + if stage.state == StageState::Running { + stage.state = StageState::Pending; + } + } + WaitState::Running | WaitState::AwaitingAnswer | WaitState::Cancelling => { + stage.state = StageState::Running; + } + WaitState::AwaitingRetry => stage.state = StageState::Retrying, + } + } + } + ViewEvent::RetryScheduled { .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.state = StageState::Retrying; + } + } + ViewEvent::VisitCompleted { + outcome, + executed, + attempts, + } => { + let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { + return; + }; + stage.state = match outcome.status { + Status::Success => StageState::Succeeded, + Status::PartialSuccess { .. } => StageState::PartiallySucceeded, + Status::Failure(_) | Status::TimedOut => StageState::Failed, + Status::Skipped => StageState::Skipped, + Status::Cancelled => StageState::Cancelled, + }; + if stage.completion.is_none() || !*executed { + stage.completion = Some(completion(&outcome.status, at)); + } + if stage.timing.is_none() { + let wall = stage + .started_at + .map_or(0, |started| timing::elapsed_ms(started, at)); + stage.set_authoritative_timing(StageTiming::new(wall, 0, 0)); + } + debug!(attempts, "visit completed"); + } + ViewEvent::ForkCompleted { + occurrence, + results, + .. + } => { + let key = FiringKey::new(occurrence.execution.raw(), occurrence.firing.raw()); + let Some(stage_id) = self + .state + .stages + .get(&key) + .map(|stage| stage.stage_id.clone()) + else { + return; + }; + let Some(projection) = self.projection.as_mut() else { + return; + }; + if let Some(stage) = projection.stage_mut(&stage_id) { + stage.parallel_results = Some( + results + .iter() + .map(|result| ParallelBranchResult { + id: result.node.name.to_string(), + index: Some(result.branch.index as usize), + item_label: None, + status: stage_outcome(&result.status), + context_updates: BTreeMap::new(), + }) + .collect(), + ); + } + } + ViewEvent::ForkStarted { .. } + | ViewEvent::BranchCompleted { .. } + | ViewEvent::RunStalled { .. } => {} + } + } + + /// A firing exists: register its stage and, when it is a logical stage, + /// A firing exists: register its stage and, when it is a logical stage, + /// show it. + pub(super) fn start_visit( + &mut self, + execution: ExecutionId, + subject: &Subject, + at: DateTime, + ) { + let Some(firing) = subject.firing else { + return; + }; + let key = FiringKey::new(execution.raw(), firing.raw()); + if self.state.stages.contains_key(&key) { + return; + } + let node_name = subject.node.name.to_string(); + let visit = visit_of(subject); + let meta_kind = node_meta_kind(&subject.node); + let shown = is_shown(&subject.node); + // Only a shown stage takes a label: a lowering node (a branch's + // parent-side delegate shares its target's name) never competes with + // the stage it stands for. + let mut stage_id = StageId::new(node_name.clone(), visit); + if shown { + stage_id = stage_label(&node_name, visit, execution, &self.state.labels); + self.state.labels.insert(stage_id.to_string()); + } + self.state.stages.insert(key, StageRef { + stage_id: stage_id.clone(), + shown, + node_name, + visit, + }); + if !shown { + return; + } + let branch = self + .state + .executions + .get(&execution.raw()) + .and_then(|invocation| self.state.invocations.get(invocation)) + .and_then(|invocation| invocation.branch.clone()); + let Some(projection) = self.projection.as_mut() else { + return; + }; + let since_created = at + .signed_duration_since(projection.spec.run_id.created_at()) + .num_milliseconds() + .max(0); + let ordinal = u32::try_from(since_created) + .unwrap_or(u32::MAX - 1) + .saturating_add(1); + let stage = projection.stage_entry(stage_id.node_id(), visit, first_event_seq(ordinal)); + stage.handler = Some(StageHandler::from_handler_type(Some(meta_kind))); + stage.started_at = Some(at); + stage.graph_visit = Some(visit); + stage.state = StageState::Pending; + stage.parallel_branch_id = branch.map(|(group, index)| ParallelBranchId::new(group, index)); + } +} + +pub(super) fn stage_outcome(status: &Status) -> StageOutcome { + match status { + Status::Success => StageOutcome::Succeeded, + Status::PartialSuccess { .. } => StageOutcome::PartiallySucceeded, + Status::Failure(info) => StageOutcome::Failed { + retry_requested: info.class.as_str() == "retry_requested", + }, + Status::Skipped => StageOutcome::Skipped, + Status::Cancelled | Status::TimedOut => StageOutcome::Failed { + retry_requested: false, + }, + } +} + +pub(super) fn failure_message(status: &Status) -> Option { + match status { + Status::Failure(info) + | Status::PartialSuccess { + underlying: Some(info), + } => Some(info.message.clone()), + Status::TimedOut => Some("the step timed out".to_string()), + Status::Cancelled => Some("the step was cancelled".to_string()), + Status::Success | Status::PartialSuccess { underlying: None } | Status::Skipped => None, + } +} + +/// A stage's completion from an attempt's status: its outcome and, for a +/// failure, the message. +fn completion(status: &Status, at: DateTime) -> StageCompletion { + StageCompletion { + outcome: stage_outcome(status), + notes: None, + failure_reason: failure_message(status), + timestamp: at, + } +} + +/// The finished attempt's metrics onto its stage: the timing and the usage +/// the backend reported. +fn apply_metrics(stage: &mut StageProjection, metrics: &Metrics) { + let custom = &metrics.custom; + let inference = custom + .get("pebble.inference_ms") + .and_then(Value::as_u64) + .unwrap_or(0); + let tool = custom + .get("pebble.tool_ms") + .and_then(Value::as_u64) + .unwrap_or(0); + let wall = metrics.duration_ms.unwrap_or(0); + let (inference, tool) = match stage.handler { + Some(StageHandler::Prompt) => (wall, 0), + Some(StageHandler::Command) => (0, wall), + _ => (inference, tool), + }; + stage.set_authoritative_timing(StageTiming::new(wall, inference, tool).clamped_to_wall()); + if let Some(usage) = + usage_of(custom.get("pebble.usage")).or_else(|| usage_of(custom.get("prompt.usage"))) + { + stage.usage = usage; + } + if let Some(sessions) = custom + .get("pebble.subagents") + .and_then(|subagents| subagents.get("sessions")) + .and_then(Value::as_array) + { + let mut by_model: Vec = Vec::new(); + for session in sessions { + let provider = session.get("provider").and_then(Value::as_str); + let model = session.get("model").and_then(Value::as_str); + let Some(usage) = usage_of(session.get("usage")) else { + continue; + }; + let Some(model) = model.and_then(|model| model_ref(provider, model)) else { + continue; + }; + if let Some(entry) = by_model.iter_mut().find(|entry| entry.model == model) { + entry.usage = entry.usage.saturating_add(usage); + } else { + by_model.push(ModelUsage::new(model, usage)); + } + } + if !by_model.is_empty() { + stage.usage_by_model = by_model; + } + } +} diff --git a/lib/components/fabro-petri/src/projection/mod.rs b/lib/components/fabro-petri/src/projection/mod.rs new file mode 100644 index 000000000..e7a4b9cd8 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/mod.rs @@ -0,0 +1,438 @@ +//! The projection of a Petri run: Petri's public events and Fabro's platform +//! records folded into the view Fabro's read side serves. +//! +//! The fold is pure. [`RunView`] holds the [`RunProjection`] the API serves +//! (`GET /runs/{id}/state`, the run list through its summary) and the +//! bookkeeping the fold needs between items ([`FoldState`]): which Petri +//! firing each stage is, which invocation each execution belongs to and +//! whether it is a parallel branch, which stage asked each open question. +//! Both halves are stored by the projector and reloaded for the next pass, +//! so a pass folds only the items past the committed positions. +//! +//! The mapping follows `VIEWS.md`, row by row. The stage key is `(execution, +//! firing)`; Fabro's `StageId` (`node@visit`) is the display label the +//! `RunProjection` keys stages by, and a label two firings would share (two +//! child invocations with the same node name and visit) is made unique by +//! naming the execution. What the matrix leaves default is left default +//! here and named in the crate's README. +//! +//! Every item the fold sees carries the delivery sequence the projector +//! assigned it (`stream_seq`), which a checkpoint keeps as its `seq`. A +//! stage's `first_event_seq`, the key the stage list sorts by, is not the +//! delivery sequence: two logs' records can be committed in an order that +//! differs from their recording times by a few positions, and the view +//! built live must equal the view rebuilt from the records alone. It is the +//! milliseconds from the run's creation to the stage's `visit.started`, +//! plus one, which is the same however the records were delivered. + +mod coordinator; +mod engine; +mod model; +mod platform; +mod progress; +mod sandbox; + +use std::collections::{BTreeMap, BTreeSet}; +use std::fmt; +use std::str::FromStr; + +use chrono::{DateTime, TimeZone as _, Utc}; +use fabro_store::StagePosition; +use fabro_store::platform_records::StoredPlatformRecord; +use fabro_types::{ + RunControlAction, RunDiff, RunId, RunProjection, RunStatus, StageId, StageProjection, +}; +use petri_execution::ExecutionId; +use petri_execution::events::{NodeRef, RunEvent, Subject}; +use serde::{Deserialize, Deserializer, Serialize, Serializer, de}; +use serde_json::Value; +use tracing::debug; + +/// One item the projector hands the fold, with its delivery sequence. +pub enum Item<'a> { + Petri(&'a RunEvent), + Platform(&'a StoredPlatformRecord), +} + +/// A stage as the fold knows it: its label in the projection, and what it +/// learned about it. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct StageRef { + pub stage_id: StageId, + /// Whether the stage is a logical one the projection shows, or a + /// lowering node it keeps off the list. + pub shown: bool, + /// The node's instance name and visit, for the collision rule. + pub node_name: String, + pub visit: u32, +} + +/// What the fold knows about one invocation. +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub struct InvocationRef { + /// The calling execution and firing, for a nested invocation. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub parent: Option<(u64, u64)>, + /// The parallel group and branch index, for a branch child. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub branch: Option<(StageId, u32)>, + /// The result the invocation recorded, for the root. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub failure: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output: Option, +} + +/// Whether the run's durable record is whole, as the projector last read it. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct RecordHealth { + pub complete: bool, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub incomplete: Vec, +} + +/// The fold's bookkeeping between items. +#[derive(Clone, Debug, Default, Serialize, Deserialize)] +pub struct FoldState { + /// Stages by firing. + #[serde(default)] + pub stages: BTreeMap, + /// Labels taken, so a second firing with the same name and visit gets + /// its own. + #[serde(default)] + pub labels: BTreeSet, + #[serde(default)] + pub invocations: BTreeMap, + /// Which invocation each execution belongs to. + #[serde(default)] + pub executions: BTreeMap, + /// Open questions by id: the firing that asked. + #[serde(default)] + pub questions: BTreeMap, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub root: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub started_at: Option, + /// The run's recorded finish, when Petri recorded one. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub finished: Option, + /// The run branch and base sha, when they arrive before `run.started`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub run_branch: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub base_sha: Option, + #[serde(default)] + pub checkpoints: u32, + /// The run's diff as its `run.diff` record gave it, whichever side of + /// the run's finish it arrived on. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub run_diff: Option, + #[serde(default)] + pub health: RecordHealth, + /// Firings whose attempt has recorded a finish: what a position-keyed + /// platform record may be streamed behind. + #[serde(default)] + pub finished_firings: BTreeSet, + /// Whether the run's sandbox still exists after its release + /// (`scope.released` `retained`): kept stopped, or deleted. Absent until + /// the root invocation's lease was released. The view carries the same + /// fact as `RunSandboxInstance.retained`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sandbox_retained: Option, +} + +impl FoldState { + /// Whether Petri recorded the run's finish. + #[must_use] + pub fn finished_run(&self) -> bool { + self.finished.is_some() + } +} + +/// The view of one run: what the API serves and what the fold keeps. +#[derive(Clone, Debug)] +pub struct RunView { + pub projection: Option, + pub state: FoldState, +} + +impl RunView { + #[must_use] + pub fn new() -> Self { + Self { + projection: None, + state: FoldState::default(), + } + } + + /// Fold one item at its delivery sequence. + pub fn fold(&mut self, item: &Item<'_>, stream_seq: u64) { + match item { + Item::Platform(record) => self.fold_platform(record, stream_seq), + Item::Petri(event) => self.fold_petri(event), + } + } + + /// The run's projection, once its `run.created` record was folded. + #[must_use] + pub fn projection(&self) -> Option<&RunProjection> { + self.projection.as_ref() + } + + fn fold_petri(&mut self, event: &RunEvent) { + let at = millis(event.recorded_at); + if let Some(record) = event.coordinator() { + self.fold_coordinator(record, event, at); + } else if let Some(engine) = event.engine() { + self.fold_engine(engine, event, at); + } else if let Some(view) = event.view() { + self.fold_view(view, event, at); + } + if let Some(projection) = self.projection.as_mut() { + touch(projection, at); + } + } + + /// The shown stage an event's subject firing belongs to. + fn stage_of( + &mut self, + execution: ExecutionId, + subject: Option<&Subject>, + ) -> Option<&mut StageProjection> { + let firing = subject?.firing?; + let stage = self + .state + .stages + .get(&FiringKey::new(execution.raw(), firing.raw()))?; + if !stage.shown { + return None; + } + let stage_id = stage.stage_id.clone(); + self.projection.as_mut()?.stage_mut(&stage_id) + } +} + +impl Default for RunView { + fn default() -> Self { + Self::new() + } +} + +// ── Shared by the folds ───────────────────────────────────────────────── + +/// Apply a status transition; one the lifecycle refuses is logged and +/// skipped, since the view never fails the run. +fn apply_status(projection: &mut RunProjection, status: RunStatus, at: DateTime) { + if let Err(error) = projection.try_apply_status(status, at) { + debug!(error = %error, "status transition not applied to the Petri projection"); + } +} + +fn touch(projection: &mut RunProjection, at: DateTime) { + if at > projection.last_event_at { + projection.last_event_at = at; + } +} + +/// A control the run acknowledged: the pending control is cleared when it +/// is the one that landed. +fn settle_control(projection: &mut RunProjection, action: RunControlAction) { + if projection.pending_control == Some(action) { + projection.pending_control = None; + } +} + +/// The key of a stage: the execution and firing of the visit it shows. The +/// same fact a positioned platform record carries as its `StagePosition`. +/// It is written `:`, which is how the stored fold +/// state keys its maps. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub struct FiringKey { + pub execution: u64, + pub firing: u64, +} + +impl FiringKey { + #[must_use] + pub fn new(execution: u64, firing: u64) -> Self { + Self { execution, firing } + } + + /// The firing an event belongs to: its context's execution and its + /// subject's firing, when it has both. + #[must_use] + pub fn of_event(event: &RunEvent) -> Option { + let execution = event.context.execution?; + let firing = event.subject.as_ref()?.firing?; + Some(Self::new(execution.raw(), firing.raw())) + } +} + +impl From for FiringKey { + fn from(position: StagePosition) -> Self { + Self::new(position.execution, position.firing) + } +} + +impl fmt::Display for FiringKey { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}:{}", self.execution, self.firing) + } +} + +/// A firing key that is not `:`. +#[derive(Debug, thiserror::Error)] +#[error("a firing key is `:`, not {0:?}")] +pub struct ParseFiringKeyError(String); + +impl FromStr for FiringKey { + type Err = ParseFiringKeyError; + + fn from_str(text: &str) -> Result { + let invalid = || ParseFiringKeyError(text.to_string()); + let (execution, firing) = text.split_once(':').ok_or_else(invalid)?; + Ok(Self::new( + execution.parse().map_err(|_| invalid())?, + firing.parse().map_err(|_| invalid())?, + )) + } +} + +impl Serialize for FiringKey { + fn serialize(&self, serializer: S) -> Result { + serializer.collect_str(self) + } +} + +impl<'de> Deserialize<'de> for FiringKey { + fn deserialize>(deserializer: D) -> Result { + String::deserialize(deserializer)? + .parse() + .map_err(de::Error::custom) + } +} + +/// Which firing of its node a subject is, 1-based. +#[must_use] +pub fn visit_of(subject: &Subject) -> u32 { + subject.visit.unwrap_or(1).max(1) +} + +/// The role a frontend gave a node under `meta.kind`, or the empty string. +fn node_meta_kind(node: &NodeRef) -> &str { + node.meta.get("kind").and_then(Value::as_str).unwrap_or("") +} + +/// Whether a node is a logical stage the projection shows, or a lowering +/// node it keeps off the list: one a frontend marked synthetic, or a +/// parallel branch's delegate. +#[must_use] +pub fn is_shown(node: &NodeRef) -> bool { + let synthetic = node + .meta + .get("synthetic") + .and_then(Value::as_bool) + .unwrap_or(false); + !synthetic && node_meta_kind(node) != "parallel.branch" +} + +/// The label a shown firing takes, which is the stage id the projection +/// keys it by: `node@visit`, or `node/e@visit` when another +/// execution's firing already took that label. `taken` is every label given +/// so far; the caller adds the one returned. The interview adapter labels a +/// question's stage through this same rule, so the stage a question names +/// is the stage the projection shows. +#[must_use] +pub fn stage_label( + node_name: &str, + visit: u32, + execution: ExecutionId, + taken: &BTreeSet, +) -> StageId { + let stage_id = StageId::new(node_name.to_string(), visit); + if taken.contains(&stage_id.to_string()) { + return StageId::new(format!("{node_name}/e{}", execution.raw()), visit); + } + stage_id +} + +fn millis(recorded_at: u64) -> DateTime { + Utc.timestamp_millis_opt(i64::try_from(recorded_at).unwrap_or(i64::MAX)) + .single() + .unwrap_or_default() +} + +/// The run id a Petri run key names. +#[must_use] +pub fn run_id_of(key: &str) -> Option { + key.parse().ok() +} + +#[cfg(test)] +mod tests { + use fabro_store::platform_records::{PlatformRecord, RunCreatedRecord}; + use fabro_types::test_support as types_support; + use petri_runtime::driver::BranchRole; + use petri_runtime::ir::{FiringId, NodeId}; + + use super::*; + + #[test] + fn a_taken_label_is_made_unique_by_the_execution() { + let mut view = RunView::new(); + let created = StoredPlatformRecord { + seq: 1, + recorded_at: 1_000, + record: PlatformRecord::RunCreated(RunCreatedRecord { + spec: types_support::test_run_spec(), + title: Some("A run".to_string()), + parent_id: None, + retried_from: None, + web_url: None, + }), + position: None, + }; + view.fold(&Item::Platform(&created), 1); + let subject = |name: &str| Subject { + node: NodeRef { + id: NodeId::new(1), + name: name.into(), + kind: "attractor/command".into(), + meta: serde_json::json!({ "kind": "command" }), + }, + firing: Some(FiringId::new(4)), + visit: Some(1), + attempt: None, + generation: None, + branch: BranchRole::None, + }; + view.start_visit(ExecutionId::new(1), &subject("build"), millis(2_000)); + view.start_visit(ExecutionId::new(2), &subject("build"), millis(3_000)); + let labels: Vec = view + .projection() + .expect("the run was created") + .iter_stages() + .map(|(id, _)| id.to_string()) + .collect(); + assert_eq!(labels, vec!["build@1", "build/e2@1"]); + assert_eq!(view.state.stages.len(), 2); + } + + #[test] + fn a_firing_key_is_stored_as_execution_colon_firing() { + let mut stages: BTreeMap = BTreeMap::new(); + stages.insert(FiringKey::new(3, 7), 1); + let json = serde_json::to_string(&stages).expect("the map encodes"); + assert_eq!(json, r#"{"3:7":1}"#); + let back: BTreeMap = serde_json::from_str(&json).expect("the map decodes"); + assert_eq!(back, stages); + assert!("3-7".parse::().is_err()); + assert_eq!( + FiringKey::from(StagePosition { + execution: 3, + firing: 7, + }), + FiringKey::new(3, 7) + ); + } +} diff --git a/lib/components/fabro-petri/src/projection/model.rs b/lib/components/fabro-petri/src/projection/model.rs new file mode 100644 index 000000000..20d24a472 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/model.rs @@ -0,0 +1,41 @@ +//! The model and usage facts the records carry, as the view names them. + +use fabro_types::ModelRef; +use lithos_llm::catalog::{ModelId, ProviderId}; +use lithos_llm::types::Usage; +use serde_json::Value; + +pub(super) fn usage_of(value: Option<&Value>) -> Option { + serde_json::from_value(value?.clone()).ok() +} + +/// `provider/model` into its parts, or the model alone. +pub(super) fn split_model(model: &str) -> (Option<&str>, &str) { + match model.split_once('/') { + Some((provider, model)) if !provider.is_empty() && !model.is_empty() => { + (Some(provider), model) + } + _ => (None, model), + } +} + +pub(super) fn model_ref(provider: Option<&str>, model: &str) -> Option { + let provider = provider.filter(|provider| !provider.is_empty())?; + Some(ModelRef::new( + ProviderId::new(provider), + ModelId::new(model), + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_model_selector_splits_into_provider_and_model() { + assert_eq!(split_model("openai/gpt-5.4"), (Some("openai"), "gpt-5.4")); + assert_eq!(split_model("gpt-5.4"), (None, "gpt-5.4")); + assert!(model_ref(None, "gpt-5.4").is_none()); + assert!(model_ref(Some("openai"), "gpt-5.4").is_some()); + } +} diff --git a/lib/components/fabro-petri/src/projection/platform.rs b/lib/components/fabro-petri/src/projection/platform.rs new file mode 100644 index 000000000..7109e25fd --- /dev/null +++ b/lib/components/fabro-petri/src/projection/platform.rs @@ -0,0 +1,319 @@ +//! The platform records folded into the view: the run's creation, its +//! lifecycle, its title and parent, the branch and identity the first +//! checkpoint recorded, every checkpoint and artifact, the run's diff, and +//! the pull request (VIEWS.md "Run", "Checkpoints", "Artifacts", "Pull +//! request"). + +use chrono::{DateTime, Utc}; +use fabro_store::platform_records::{ + PlatformRecord, RunLifecycleKind, RunLifecycleRecord, StoredPlatformRecord, +}; +use fabro_types::{ + CheckpointRecord as ViewCheckpoint, Conclusion, FailureCategory, FailureDetail, FailureReason, + PullRequestCreation, PullRequestCreationStatus, PullRequestLink, RunApproval, RunApprovalState, + RunArtifact, RunControlAction, RunDiff, RunFailure, RunProjection, RunSandbox, RunStatus, + StageOutcome, format_blob_ref, +}; +use tracing::debug; + +use super::sandbox::sandbox_plan; +use super::{FiringKey, RunView, apply_status, millis, settle_control, touch}; + +impl RunView { + pub(super) fn fold_platform(&mut self, stored: &StoredPlatformRecord, stream_seq: u64) { + let at = millis(stored.recorded_at); + if let PlatformRecord::RunCreated(created) = &stored.record { + let title = created + .title + .clone() + .unwrap_or_else(|| fabro_types::infer_run_title(created.spec.graph.goal())); + let mut projection = RunProjection::new(title, created.spec.clone(), at); + projection.parent_id = created.parent_id; + projection.retried_from = created.retried_from; + projection.web_url.clone_from(&created.web_url); + projection.sandbox = Some(RunSandbox::planned(sandbox_plan( + &projection.spec.settings.run.environment, + ))); + self.projection = Some(projection); + return; + } + let Some(projection) = self.projection.as_mut() else { + debug!( + seq = stored.seq, + kind = %stored.record.kind(), + "platform record before run.created; not folded" + ); + return; + }; + touch(projection, at); + match &stored.record { + PlatformRecord::RunLifecycle(record) => fold_lifecycle(projection, record, at), + PlatformRecord::RunTitle(record) => projection.title.clone_from(&record.title), + PlatformRecord::RunParent(record) => projection.parent_id = record.parent_id, + PlatformRecord::RunArchived => projection.archived_at = Some(at), + PlatformRecord::RunUnarchived => projection.archived_at = None, + PlatformRecord::RunSuperseded(record) => { + projection.superseded_by = Some(record.new_run_id); + } + PlatformRecord::RunCreated(_) + | PlatformRecord::RunNotice(_) + | PlatformRecord::InterviewAnswered(_) + | PlatformRecord::NotificationSent(_) + | PlatformRecord::RunPaired(_) => {} + PlatformRecord::RunBranch(record) => { + self.state.run_branch.clone_from(&record.run_branch); + self.state.base_sha.clone_from(&record.base_sha); + if let Some(start) = projection.start.as_mut() { + start.run_branch.clone_from(&record.run_branch); + start.base_sha.clone_from(&record.base_sha); + } + } + PlatformRecord::GitIdentity(record) => { + projection.git_identity = Some(record.identity.clone()); + } + PlatformRecord::Checkpoint(record) => { + self.state.checkpoints = self.state.checkpoints.saturating_add(1); + let stage = self + .state + .stages + .get(&FiringKey::new(record.execution, record.firing)); + let current_node = stage.map_or_else(String::new, |stage| stage.node_name.clone()); + let stage_id = stage + .filter(|stage| stage.shown) + .map(|stage| stage.stage_id.clone()); + let checkpoint = fabro_types::Checkpoint { + timestamp: at, + current_node: current_node.clone(), + git_commit_sha: record.git_commit_sha.clone(), + }; + // The patch stays in the blob table; the view carries its + // reference for a reader to resolve. + let patch = record.patch_blob.as_ref().map(format_blob_ref); + if let Some(stage) = stage_id.and_then(|stage_id| projection.stage_mut(&stage_id)) { + if patch.is_some() { + stage.diff.clone_from(&patch); + } + } + projection.checkpoints.push(ViewCheckpoint { + seq: u32::try_from(stream_seq).unwrap_or(u32::MAX), + checkpoint, + diff: RunDiff { + patch, + summary: record.diff_summary, + }, + }); + } + PlatformRecord::ArtifactCollected(record) => { + let stage = self + .state + .stages + .get(&FiringKey::new(record.execution, record.firing)); + let Some(stage_id) = stage.map(|stage| stage.stage_id.clone()) else { + debug!( + seq = stored.seq, + path = record.path, + "artifact record for an unknown firing; not folded" + ); + return; + }; + projection.artifacts.push(RunArtifact { + stage_id, + retry: record.attempt, + relative_path: record.path.clone(), + size: record.bytes, + blob: record.blob, + }); + } + PlatformRecord::RunDiff(record) => { + let diff = RunDiff { + patch: record.patch_blob.as_ref().map(format_blob_ref), + summary: record.diff_summary, + }; + if let Some(conclusion) = projection.conclusion.as_mut() { + conclusion.diff = diff.clone(); + } + self.state.run_diff = Some(diff); + } + PlatformRecord::PullRequestRequested(record) => { + projection.pull_request_creation = Some(PullRequestCreation { + id: record.creation_id, + status: PullRequestCreationStatus::Pending, + model: record.model.clone(), + force: record.force, + requested_at: at, + updated_at: at, + pull_request: None, + error: None, + }); + } + PlatformRecord::PullRequestCreated(record) => { + let link = PullRequestLink { + owner: record.owner.clone(), + repo: record.repo.clone(), + number: record.number, + }; + projection.pull_request = Some(link.clone()); + if let Some(creation) = projection + .pull_request_creation + .as_mut() + .filter(|creation| creation.is_pending()) + { + creation.succeed(link, at); + } + } + PlatformRecord::PullRequestFailed(record) => { + if let Some(creation) = + projection + .pull_request_creation + .as_mut() + .filter(|creation| { + creation.is_pending() + && record + .creation_id + .is_none_or(|creation_id| creation_id == creation.id) + }) + { + creation.fail(record.error.clone(), at); + } + } + PlatformRecord::PullRequestLinked(record) => { + let link = record.link(); + projection.pull_request = Some(link.clone()); + if let Some(creation) = projection + .pull_request_creation + .as_mut() + .filter(|creation| creation.is_pending()) + { + creation.succeed(link, at); + } + } + PlatformRecord::PullRequestUnlinked(_) => { + projection.pull_request = None; + projection.pull_request_creation = None; + } + } + } +} + +fn fold_lifecycle(projection: &mut RunProjection, record: &RunLifecycleRecord, at: DateTime) { + use RunLifecycleKind as Kind; + match record.transition { + Kind::Submitted => apply_status(projection, RunStatus::Submitted, at), + Kind::StartRequested => {} + Kind::Unpaused => { + apply_status(projection, projection.status.unpaused(), at); + settle_control(projection, RunControlAction::Unpause); + } + Kind::Pending => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + projection.approval = Some(RunApproval { + state: RunApprovalState::Pending, + requested_at: at, + decided_at: None, + denial_reason: None, + }); + } + Kind::Approved => { + if let Some(approval) = projection.approval.as_mut() { + approval.state = RunApprovalState::Approved; + approval.decided_at = Some(at); + } + } + Kind::Denied => { + if let Some(approval) = projection.approval.as_mut() { + approval.state = RunApprovalState::Denied; + approval.decided_at = Some(at); + approval.denial_reason.clone_from(&record.reason); + } + apply_status( + projection, + RunStatus::Failed { + reason: FailureReason::ApprovalDenied, + }, + at, + ); + } + Kind::Runnable => { + // A run left in flight by a restart goes back to the queue: the + // resume's `runnable` steps back from wherever the run stood. + let in_flight = matches!( + projection.status, + RunStatus::Starting + | RunStatus::Running + | RunStatus::Blocked { .. } + | RunStatus::Paused { .. } + ); + if in_flight && record.status == Some(RunStatus::Runnable) { + projection.status = RunStatus::Runnable; + projection.status_updated_at = at; + } else if let Some(status) = record.status { + apply_status(projection, status, at); + } + } + Kind::Blocked => { + // A block that lands while the run is paused waits behind the + // pause: the unpause restores it. + match (projection.status, record.status) { + (RunStatus::Paused { .. }, Some(RunStatus::Blocked { blocked_reason })) => { + apply_status( + projection, + RunStatus::Paused { + prior_block: Some(blocked_reason), + }, + at, + ); + } + (_, Some(status)) => apply_status(projection, status, at), + (_, None) => {} + } + } + Kind::Unblocked => { + let status = match projection.status { + RunStatus::Paused { .. } => RunStatus::Paused { prior_block: None }, + _ => record.status.unwrap_or(RunStatus::Running), + }; + apply_status(projection, status, at); + } + Kind::Starting | Kind::Running | Kind::Removing | Kind::Dead => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + } + Kind::Paused => { + apply_status(projection, projection.status.paused(), at); + settle_control(projection, RunControlAction::Pause); + } + Kind::Succeeded | Kind::Failed => { + if let Some(status) = record.status { + apply_status(projection, status, at); + } + projection.pending_control = None; + if projection.conclusion.is_none() { + let (outcome, failure) = match record.status { + Some(RunStatus::Failed { reason }) => ( + StageOutcome::Failed { + retry_requested: false, + }, + Some(RunFailure { + reason, + detail: FailureDetail::new( + record + .reason + .clone() + .unwrap_or_else(|| "the run failed".to_string()), + FailureCategory::Deterministic, + ), + }), + ), + _ => (StageOutcome::Succeeded, None), + }; + projection.conclusion = Some(Conclusion::outcome_only(at, outcome, failure)); + } + } + Kind::CancelRequested => projection.pending_control = Some(RunControlAction::Cancel), + Kind::PauseRequested => projection.pending_control = Some(RunControlAction::Pause), + Kind::UnpauseRequested => projection.pending_control = Some(RunControlAction::Unpause), + } +} diff --git a/lib/components/fabro-petri/src/projection/progress.rs b/lib/components/fabro-petri/src/projection/progress.rs new file mode 100644 index 000000000..ad7cc6a84 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/progress.rs @@ -0,0 +1,573 @@ +//! A step's progress records folded into its stage: a command's log lines, +//! the payloads the Attractor steps emit (the prompt and its completion, +//! the fallback plan, the tools a session was offered, a parallel branch's +//! start), Pebble's coding-agent envelope, and a gate's question (VIEWS.md +//! "Agent activity", "Questions"). + +use chrono::{DateTime, Utc}; +use fabro_types::{ + BlockedReason, CodingAgentEvent, CodingEvent, InterviewOption, InterviewQuestionRecord, + PendingInterviewRecord, ReviewTarget, ReviewTargetKind, RunStatus, StageInferenceProjection, + StageModelUsage, StageProjection, ToolCategory, ToolSource, ToolSummary, timing, +}; +use lithos_llm::types::{ReasoningEffort, Speed, Usage}; +use petri_execution::ExecutionId; +use petri_execution::events::{Parsed, RunEvent}; +use petri_runtime::ir::StepEvent; +use petri_runtime::steps::QuestionReference; +use serde::Deserialize; +use serde_json::Value; +use tracing::debug; + +use super::model::{model_ref, split_model}; +use super::{FiringKey, RunView, apply_status}; +use crate::interview::question_type; + +impl RunView { + pub(super) fn fold_progress( + &mut self, + execution: ExecutionId, + event: &RunEvent, + ev: &StepEvent, + at: DateTime, + ) { + match ev { + StepEvent::Log { line, .. } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + let output = stage.output.get_or_insert_default(); + output.push_str(line); + output.push('\n'); + stage.output_bytes = Some(output.len() as u64); + stage.live_streaming = Some(true); + } + } + StepEvent::Artifact { .. } => {} + StepEvent::Custom(payload) => { + if let Some(parsed) = event.parsed() { + self.fold_parsed(execution, event, parsed, at); + return; + } + let progress = match Progress::deserialize(payload) { + Ok(progress) => progress, + Err(error) => { + debug!(error = %error, "a progress payload did not decode; skipped"); + return; + } + }; + match progress { + Progress::Pebble { event: envelope } => { + self.fold_pebble(execution, event, &envelope, at); + } + Progress::Prompt { prompt, model } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.prompt = prompt; + if let Some(model) = model.as_deref() { + let (provider, model_id) = split_model(model); + stage.provider_used = Some(StageModelUsage::new( + StageModelUsage::MODE_PROMPT, + provider.map(str::to_string), + Some(model_id.to_string()), + )); + stage.model = model_ref(provider, model_id); + } + } + } + Progress::PromptCompleted { response, usage } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + stage.response = response; + if let Some(usage) = usage { + stage.usage = usage; + } + } + } + // `routes[0]` is the original route: what the stage was + // asked to run on, with its request controls. + Progress::FallbackPlan { routes } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + if let Some(route) = routes.into_iter().next() { + stage.model = model_ref(Some(&route.provider), &route.model); + stage.provider_used = Some(route.usage()); + } + } + } + // The tools a native session was offered, once per + // session (VIEWS.md "Agent activity", tools available): + // the stage's list is the union over its sessions, by + // name, in the order the sessions listed them. + Progress::Tools { tools } => { + if let Some(stage) = self.stage_of(execution, event.subject.as_ref()) { + for tool in tools { + if !stage + .agent_tools + .iter() + .any(|known| known.name == tool.name) + { + stage.agent_tools.push(tool.summary()); + } + } + } + } + Progress::BranchStarted { + invocation, + index, + occurrence, + } => { + let group = self + .state + .stages + .get(&FiringKey::new(execution.raw(), occurrence.firing)) + .map(|stage| stage.stage_id.clone()); + if let Some(group) = group { + self.state.invocations.entry(invocation).or_default().branch = + Some((group, index)); + } + } + Progress::Other => {} + } + } + } + } + + fn fold_parsed( + &mut self, + execution: ExecutionId, + event: &RunEvent, + parsed: &Parsed, + at: DateTime, + ) { + match parsed { + Parsed::Question { question } => { + let Some(subject) = event.subject.as_ref() else { + return; + }; + let Some(firing) = subject.firing else { + return; + }; + let key = FiringKey::new(execution.raw(), firing.raw()); + let label = self.state.stages.get(&key).map_or_else( + || subject.node.name.to_string(), + |stage| stage.stage_id.to_string(), + ); + self.state.questions.insert(question.id.clone(), key); + let Some(projection) = self.projection.as_mut() else { + return; + }; + projection + .pending_interviews + .insert(question.id.clone(), PendingInterviewRecord { + question: InterviewQuestionRecord { + id: question.id.clone(), + text: question.text.clone(), + stage: label, + question_type: question_type(question), + options: question + .options + .iter() + .map(|option| InterviewOption { + key: option.key.clone(), + label: option.label.clone(), + description: option.description.clone(), + preview: option.preview.clone(), + }) + .collect(), + allow_freeform: question.freeform, + timeout_seconds: question + .timeout_ms + .map(|timeout| timeout as f64 / 1000.0), + context_display: question.context.clone(), + review_target: question.reference.as_ref().and_then(review_target), + }, + started_at: at, + }); + apply_status( + projection, + RunStatus::Blocked { + blocked_reason: BlockedReason::HumanInputRequired, + }, + at, + ); + } + Parsed::QuestionExpired { expired } => { + self.close_questions(Some(expired.question.as_str()), None, at); + } + Parsed::Note { .. } => {} + } + } + + /// Close one question by id, or every question of a firing, and unblock + /// Close one question by id, or every question of a firing, and unblock + /// the run when none is left. + pub(super) fn close_questions( + &mut self, + question: Option<&str>, + firing: Option, + at: DateTime, + ) { + let closed: Vec = match (question, firing) { + (Some(question), _) => vec![question.to_string()], + (None, Some(firing)) => self + .state + .questions + .iter() + .filter(|(_, asked_by)| **asked_by == firing) + .map(|(id, _)| id.clone()) + .collect(), + (None, None) => Vec::new(), + }; + for id in &closed { + self.state.questions.remove(id); + } + let Some(projection) = self.projection.as_mut() else { + return; + }; + for id in &closed { + projection.pending_interviews.remove(id); + } + if projection.pending_interviews.is_empty() + && matches!(projection.status, RunStatus::Blocked { .. }) + { + apply_status(projection, RunStatus::Running, at); + } + } + + fn fold_pebble( + &mut self, + execution: ExecutionId, + event: &RunEvent, + envelope: &CodingAgentEvent, + at: DateTime, + ) { + let Some(stage) = self.stage_of(execution, event.subject.as_ref()) else { + return; + }; + let agent = stage.agent.get_or_insert_default(); + agent.apply(envelope); + if stage.completion.is_none() { + stage.usage = agent.usage.saturating_add(agent.descendant_usage()); + } + // A tool the stage's list names was called, by any of its sessions. + if let CodingEvent::ToolCallStarted { tool_name, .. } = &envelope.event { + if let Some(tool) = stage + .agent_tools + .iter_mut() + .find(|tool| tool.name == *tool_name) + { + tool.invoked = true; + } + } + let is_root = envelope.parent_session_id.is_none(); + #[expect( + clippy::wildcard_enum_match_arm, + reason = "pebble's event vocabulary is non-exhaustive and only some events project" + )] + match &envelope.event { + CodingEvent::SessionStarted { + provider, model, .. + } if is_root => { + stage.provider_used = Some(StageModelUsage::new( + StageModelUsage::MODE_AGENT, + provider.clone(), + model.clone(), + )); + if let Some(model) = model.as_deref() { + stage.model = model_ref(provider.as_deref(), model); + } + } + CodingEvent::LlmRequestStarted { requested_model } if is_root => { + stage.inference = Some(StageInferenceProjection { + session_id: envelope.session_id.clone(), + started_at: at, + requested_model: requested_model.clone(), + first_output_at: None, + first_output_kind: None, + retries: 0, + }); + } + CodingEvent::LlmFirstOutput { kind } => { + if let Some(inference) = stage.inference.as_mut() { + if inference.session_id == envelope.session_id { + inference.first_output_at = Some(at); + inference.first_output_kind = Some(*kind); + } + } + } + CodingEvent::LlmRetry { .. } => { + if let Some(inference) = stage.inference.as_mut() { + if inference.session_id == envelope.session_id { + inference.retries = inference.retries.saturating_add(1); + inference.first_output_at = None; + inference.first_output_kind = None; + } + } + } + CodingEvent::AssistantMessage { model, .. } => { + if is_root { + if let Some(provider) = stage + .provider_used + .as_ref() + .and_then(|used| used.provider.as_deref()) + { + stage.model = model_ref(Some(provider), model); + } + } + close_inference(stage, &envelope.session_id, at); + } + CodingEvent::Error { .. } | CodingEvent::RoundInterrupted { .. } => { + close_inference(stage, &envelope.session_id, at); + } + CodingEvent::SessionEnded => { + close_inference(stage, &envelope.session_id, at); + stage.close_tool_batch_for_session(&envelope.session_id, at); + } + CodingEvent::ToolCallStarted { tool_call_id, .. } if is_root => { + stage.open_tool_call(envelope.session_id.clone(), tool_call_id.clone(), at); + } + CodingEvent::ToolCallCompleted { tool_call_id, .. } if is_root => { + stage.close_tool_call(&envelope.session_id, tool_call_id, at); + } + _ => {} + } + } +} + +/// The `StepEvent::Custom` payloads the fold reads, by their `kind`. The +/// kinds are Petri's, and a test holds each literal to the constant the +/// Attractor steps export, so a rename there fails here rather than +/// projecting nothing. A payload of another kind, or of no kind, is +/// `Other`. +#[derive(Deserialize)] +#[serde(tag = "kind")] +enum Progress { + /// Pebble's coding-agent envelope, forwarded by the agent step. + #[serde(rename = "pebble")] + Pebble { event: Box }, + /// The prompt step before its first model call: the prompt, and the + /// `provider/model` selector it runs on. + #[serde(rename = "attractor.prompt")] + Prompt { + #[serde(default)] + prompt: Option, + #[serde(default)] + model: Option, + }, + /// The prompt step after its last model call. + #[serde(rename = "attractor.prompt.completed")] + PromptCompleted { + #[serde(default)] + response: Option, + #[serde(default)] + usage: Option, + }, + /// A stage's fallback plan, once per stage: `routes[0]` is the + /// original route. + #[serde(rename = "attractor.fallback.plan")] + FallbackPlan { + #[serde(default)] + routes: Vec, + }, + /// The tools a native session was offered, once per session. + #[serde(rename = "attractor.tools")] + Tools { + #[serde(default)] + tools: Vec, + }, + /// A parallel branch's child started: which fork visit it belongs to, + /// its index, and the child invocation. + #[serde(rename = "attractor.parallel.branch.started")] + BranchStarted { + invocation: u64, + index: u32, + occurrence: ForkOccurrence, + }, + #[serde(other)] + Other, +} + +/// One route of a fallback plan: the provider and model, with the request +/// controls the route carries. +#[derive(Deserialize)] +struct PlannedRoute { + provider: String, + model: String, + #[serde(default)] + reasoning_effort: Option, + #[serde(default)] + speed: Option, +} + +impl PlannedRoute { + /// The route as the stage's model usage: an agent route with its + /// controls. + fn usage(self) -> StageModelUsage { + StageModelUsage { + reasoning_effort: self.reasoning_effort, + speed: self.speed, + ..StageModelUsage::new( + StageModelUsage::MODE_AGENT, + Some(self.provider), + Some(self.model), + ) + } + } +} + +/// The fork visit a branch belongs to: the fork step's firing in the +/// branch's execution. +#[derive(Deserialize)] +struct ForkOccurrence { + firing: u64, +} + +/// One tool of an `attractor.tools` payload: the name and description as +/// recorded, Pebble's `source` as it is, and Petri's origin category +/// (`builtin`, `mcp`, `host`, `question`, `subagent`). +#[derive(Deserialize)] +struct OfferedTool { + name: String, + #[serde(default)] + description: String, + /// Left as recorded: a source Pebble adds later still lists the tool, + /// under the default source. + #[serde(default)] + source: Value, + #[serde(default)] + category: Option, +} + +impl OfferedTool { + /// The tool as the stage's list carries it. Pebble's behavioural + /// category is kept where Petri's says which (a sub-agent tool); every + /// other tool is `other`, because the payload carries Petri's origin + /// category, not Pebble's permission class. `invoked` starts false and + /// flips on the session's `ToolCallStarted`. + fn summary(self) -> ToolSummary { + let category = match self.category.as_deref() { + Some("subagent") => ToolCategory::Subagent, + _ => ToolCategory::Other, + }; + ToolSummary { + name: self.name, + description: self.description, + source: serde_json::from_value::(self.source).unwrap_or_default(), + category, + invoked: false, + } + } +} + +/// The question's `reference` as Fabro's review target, when it is one +/// Fabro's validation admits (a `document`, or a reference without a kind, +/// with a label and an absolute HTTP URL within Fabro's limits). +fn review_target(reference: &QuestionReference) -> Option { + let kind = match reference.kind.as_deref() { + Some("document") | None => ReviewTargetKind::Document, + Some(_) => return None, + }; + ReviewTarget::new(reference.label.clone(), reference.url.clone(), kind).ok() +} + +fn close_inference(stage: &mut StageProjection, session_id: &str, at: DateTime) { + let open = stage + .inference + .as_ref() + .is_some_and(|inference| inference.session_id == session_id); + if !open { + return; + } + if let Some(inference) = stage.inference.take() { + stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, at)); + } +} + +#[cfg(test)] +mod tests { + use petri_attractor_steps::{fallback, parallel, pebble, prompt}; + use serde_json::json; + + use super::*; + + /// The kinds the fold matches are the ones the Attractor steps emit. + #[test] + fn the_progress_kinds_are_petris() { + let kind_of = |value: Value| -> &'static str { + match Progress::deserialize(&value).expect("a known kind decodes") { + Progress::Pebble { .. } => "pebble", + Progress::Prompt { .. } => "prompt", + Progress::PromptCompleted { .. } => "prompt.completed", + Progress::FallbackPlan { .. } => "fallback.plan", + Progress::Tools { .. } => "tools", + Progress::BranchStarted { .. } => "branch.started", + Progress::Other => "other", + } + }; + assert_eq!(kind_of(json!({ "kind": prompt::PROMPT_EVENT })), "prompt"); + assert_eq!( + kind_of(json!({ "kind": prompt::COMPLETED_EVENT })), + "prompt.completed" + ); + assert_eq!( + kind_of(json!({ "kind": fallback::PLAN_EVENT })), + "fallback.plan" + ); + assert_eq!(kind_of(json!({ "kind": pebble::tools::EVENT })), "tools"); + assert_eq!( + kind_of(json!({ + "kind": parallel::BRANCH_STARTED_EVENT, + "invocation": 3, + "index": 1, + "occurrence": { "fork": "fan", "firing": 7 }, + })), + "branch.started" + ); + assert_eq!( + kind_of(json!({ "kind": parallel::BRANCH_COMPLETED_EVENT })), + "other" + ); + } + + /// A plan's original route carries its request controls onto the + /// stage's model usage. + #[test] + fn a_fallback_plan_route_keeps_its_controls() { + let progress = Progress::deserialize(&json!({ + "kind": fallback::PLAN_EVENT, + "routes": [ + { "position": 0, "provider": "openai", "model": "gpt-5.4", + "reasoning_effort": "high", "speed": null }, + { "position": 1, "provider": "anthropic", "model": "claude" }, + ], + })) + .expect("the plan decodes"); + let Progress::FallbackPlan { routes } = progress else { + panic!("not a plan"); + }; + let usage = routes.into_iter().next().expect("a route").usage(); + assert_eq!(usage.mode, StageModelUsage::MODE_AGENT); + assert_eq!(usage.provider.as_deref(), Some("openai")); + assert_eq!(usage.model.as_deref(), Some("gpt-5.4")); + assert_eq!(usage.reasoning_effort, Some(ReasoningEffort::High)); + assert_eq!(usage.speed, None); + } + + /// A tool lists under Petri's category, with its source as recorded. + #[test] + fn an_offered_tool_maps_to_its_summary() { + let progress = Progress::deserialize(&json!({ + "kind": pebble::tools::EVENT, + "tools": [ + { "name": "spawn_agent", "description": "a child", "source": "native", + "category": "subagent" }, + { "name": "Read", "category": "builtin" }, + ], + })) + .expect("the tools decode"); + let Progress::Tools { tools } = progress else { + panic!("not a tool list"); + }; + let summaries: Vec = tools.into_iter().map(OfferedTool::summary).collect(); + assert_eq!(summaries[0].category, ToolCategory::Subagent); + assert_eq!(summaries[1].category, ToolCategory::Other); + assert_eq!(summaries[1].description, ""); + assert!(!summaries[0].invoked); + } +} diff --git a/lib/components/fabro-petri/src/projection/sandbox.rs b/lib/components/fabro-petri/src/projection/sandbox.rs new file mode 100644 index 000000000..61ee6dca9 --- /dev/null +++ b/lib/components/fabro-petri/src/projection/sandbox.rs @@ -0,0 +1,76 @@ +//! The run's sandbox as the view shows it: the plan its settings give, and +//! the instance Petri's scope records name (VIEWS.md "Sandbox"). + +use fabro_types::settings::run::RunEnvironmentSettings; +use fabro_types::{ + RunProjection, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, SandboxProviderKind, +}; +use petri_runtime::ir::SandboxInstance; + +pub(super) fn sandbox_plan(settings: &RunEnvironmentSettings) -> RunSandboxPlan { + RunSandboxPlan { + provider: settings.provider.clone(), + image: (settings.provider == SandboxProviderKind::DOCKER) + .then(|| settings.image.docker.clone()) + .flatten() + .filter(|image| !image.is_empty()), + snapshot: None, + } +} + +/// The plan the projection's sandbox carries, or the one its environment +/// settings give when no sandbox was projected yet. +pub(super) fn sandbox_plan_of(projection: &RunProjection) -> RunSandboxPlan { + projection.sandbox.as_ref().map_or_else( + || sandbox_plan(&projection.spec.settings.run.environment), + |sandbox| sandbox.plan().clone(), + ) +} + +/// Fabro's name for the provider Petri's `scope.acquired` names: Petri's +/// `host` is Fabro's `local`; every other kind is spelled the same. `None` +/// for a name that is no provider kind. +pub(super) fn provider_kind(provider: &str) -> Option { + if provider == "host" { + return Some(SandboxProviderKind::LOCAL); + } + SandboxProviderKind::try_new(provider).ok() +} + +/// The run's sandbox instance from Petri's record of the scope's +/// acquisition: the provider, the provider's id for the sandbox (what a +/// reconnect attaches by), its image and snapshot when the provider knows +/// them, the working directory, and how long the acquisition took. The +/// clone fields stay unset: Petri's checkout copies the bound repository +/// into the workspace and is not a clone Fabro made, and the workspace +/// roots are the provider's own layout, read live. `retained` waits for +/// the scope's release. +pub(super) fn sandbox_instance( + plan: &RunSandboxPlan, + sandbox: &SandboxInstance, + ready_duration_ms: u64, +) -> RunSandboxInstance { + RunSandboxInstance { + provider: provider_kind(&sandbox.provider) + .unwrap_or_else(|| plan.provider.clone()), + image: sandbox + .image + .as_ref() + .map(ToString::to_string) + .or_else(|| plan.image.clone()), + snapshot: sandbox.snapshot.as_ref().map(ToString::to_string), + runtime: RunSandboxRuntime { + id: sandbox.instance.to_string(), + working_directory: sandbox.working_directory.to_string(), + repo_cloned: None, + clone_origin_url: None, + clone_branch: None, + workspace_root: None, + repos_root: None, + primary_repo_path: None, + primary_repo_link: None, + }, + ready_duration_ms: Some(ready_duration_ms), + retained: None, + } +} diff --git a/lib/components/fabro-petri/src/projector/cache.rs b/lib/components/fabro-petri/src/projector/cache.rs index 3791b0ced..abe4cf5fd 100644 --- a/lib/components/fabro-petri/src/projector/cache.rs +++ b/lib/components/fabro-petri/src/projector/cache.rs @@ -15,10 +15,11 @@ //! rows: the rebuild test in `tests/projection.rs` compares the two. use std::collections::HashMap; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; use fabro_types::RunId; +use fabro_util::sync; use petri_execution::events::{RunEvent, RunReplay}; use tokio::sync::Mutex as AsyncMutex; @@ -86,7 +87,7 @@ impl Caches { /// The run's pass lock, holding its cache if one is kept; the run counts /// as used now. pub(super) fn pass_of(&self, run_id: RunId) -> Arc>> { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let entry = runs.entry(run_id).or_insert_with(|| Entry { pass: Arc::default(), touched: Instant::now(), @@ -99,7 +100,7 @@ impl Caches { /// with no cache and no pass under way. A run whose pass is running is /// in use and left alone. How many caches were dropped. pub(crate) fn sweep(&self, idle: Duration) -> usize { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let mut dropped = 0; runs.retain(|_, entry| { if entry.touched.elapsed() < idle { @@ -120,13 +121,10 @@ impl Caches { } /// Whether a cache is kept for the run: a test's view of the cache. + #[cfg(any(test, feature = "test-support"))] pub(crate) fn holds(&self, run_id: RunId) -> bool { - let runs = lock(&self.runs); + let runs = sync::lock(&self.runs); runs.get(&run_id) .is_some_and(|entry| entry.pass.try_lock().is_ok_and(|cache| cache.is_some())) } } - -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} diff --git a/lib/components/fabro-petri/src/projector.rs b/lib/components/fabro-petri/src/projector/mod.rs similarity index 51% rename from lib/components/fabro-petri/src/projector.rs rename to lib/components/fabro-petri/src/projector/mod.rs index 2dea3dcf0..cada3395c 100644 --- a/lib/components/fabro-petri/src/projector.rs +++ b/lib/components/fabro-petri/src/projector/mod.rs @@ -54,20 +54,24 @@ //! completeness once the run has recorded its finish. mod cache; +pub(crate) mod order; +mod signalling; +pub(crate) mod stream; -use std::collections::{BTreeMap, BTreeSet, HashMap}; +use std::collections::{BTreeMap, HashMap}; +#[cfg(any(test, feature = "test-support"))] use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError}; +use std::sync::{Arc, Mutex}; use std::time::Duration; use fabro_db::DbPool; -use fabro_store::platform_records::{PlatformRecordStore, StoredPlatformRecord, now_ms}; +use fabro_store::platform_records::{PlatformRecordStore, now_ms}; use fabro_store::{RunProjection, RunSummaryStore}; -use fabro_types::{RunId, RunStreamItem, RunStreamItemKind}; +use fabro_types::{RunId, RunStreamItem}; use fabro_util::error::collect_chain; -use petri_execution::events::{self, EventId, EventSource, RunEvent}; +use fabro_util::sync; +use petri_execution::events::{EventId, EventSource, RunEvent}; use petri_execution::{Access, CoordinatorEvent, RunKey, RunStore as _, inspect}; -use petri_runtime::engine::Event; use petri_store::StoreError; use serde::{Deserialize, Serialize}; use tokio::sync::broadcast; @@ -75,8 +79,9 @@ use tokio::time; use tracing::{debug, info, warn}; use self::cache::{Caches, IDLE, RunCache}; +use self::stream::StreamRow; use crate::SqliteRunStore; -use crate::projection::{self, FoldState, Item, RecordHealth, RunView}; +use crate::projection::{self, FoldState, RecordHealth, RunView}; /// The positions a view committed: the last event consumed per Petri log, /// and the last platform record consumed. @@ -127,6 +132,48 @@ pub struct PassReport { pub health: RecordHealth, } +impl PassReport { + /// A pass that found the view at the head of every log and wrote + /// nothing. + fn skipped(run_id: RunId, stored: &StoredView) -> Self { + Self { + run_id, + skipped: true, + contended: false, + petri_events: 0, + platform_records: 0, + replayed_records: 0, + stream_seq: stored.stream_seq, + positions: stored.positions.clone(), + health: stored.view.state.health.clone(), + } + } + + /// A pass that left the view alone because a platform record landed + /// under it; the projector runs it again. + fn contended(run_id: RunId, replayed_records: usize) -> Self { + Self { + run_id, + skipped: false, + contended: true, + petri_events: 0, + platform_records: 0, + replayed_records, + stream_seq: 0, + positions: Positions::default(), + health: RecordHealth::default(), + } + } +} + +/// What a pass read past its cache: the events new to the view, the +/// replay failure that held it, and how many records the replay cost. +struct NewEvents { + events: Vec, + replay_failure: Option, + replayed_records: usize, +} + /// What the startup pass did. #[derive(Clone, Debug, Default, PartialEq, Eq)] pub struct StartupReport { @@ -181,8 +228,9 @@ pub struct Projector { /// over the same run never interleave their reads and writes), and the /// cache each live run's passes continue from. pub(crate) caches: Caches, - /// Test-only: stop the next pass after its reads, before its view - /// transaction, as a crash there would. + /// A test's fault: stop the next pass after its reads, before its + /// view transaction, as a crash there would. + #[cfg(any(test, feature = "test-support"))] fault: AtomicBool, /// Sent after each committed pass that wrote stream rows: the run whose /// stream grew. A wake-up for the stream's readers, never a source of @@ -209,6 +257,7 @@ impl Projector { pool: views, slots: Mutex::default(), caches: Caches::default(), + #[cfg(any(test, feature = "test-support"))] fault: AtomicBool::new(false), committed: broadcast::channel(COMMIT_SIGNAL_CAPACITY).0, }) @@ -231,7 +280,7 @@ impl Projector { after: u64, limit: usize, ) -> Result, ProjectError> { - stream_after(&self.pool, run_id, after, limit).await + stream::stream_after(&self.pool, run_id, after, limit).await } /// Delete everything the store and the view tables hold for the run: @@ -270,7 +319,7 @@ impl Projector { .map_err(ProjectError::Database)?; } records.commit().await.map_err(ProjectError::Database)?; - lock(&self.slots).remove(&run_id); + sync::lock(&self.slots).remove(&run_id); Ok(()) } @@ -290,7 +339,7 @@ impl Projector { /// more when it ends; any number of signals in between coalesce. pub fn signal(self: &Arc, run_id: RunId) { { - let mut slots = lock(&self.slots); + let mut slots = sync::lock(&self.slots); let slot = slots.entry(run_id).or_default(); if slot.running { slot.pending = true; @@ -312,7 +361,7 @@ impl Projector { false } }; - let mut slots = lock(&projector.slots); + let mut slots = sync::lock(&projector.slots); let slot = slots.entry(run_id).or_default(); if again || slot.pending { slot.pending = false; @@ -329,7 +378,7 @@ impl Projector { pub async fn settle(&self, run_id: RunId) { loop { let idle = { - let slots = lock(&self.slots); + let slots = sync::lock(&self.slots); slots .get(&run_id) .is_none_or(|slot| !slot.running && !slot.pending) @@ -343,10 +392,24 @@ impl Projector { /// Stop the next pass after its reads and before its view transaction, /// as a crash there would, once. + #[cfg(any(test, feature = "test-support"))] pub fn fail_before_view(&self) { self.fault.store(true, Ordering::SeqCst); } + /// Whether a test asked this pass to stop before its view transaction; + /// never outside tests. + fn take_fault(&self) -> bool { + #[cfg(any(test, feature = "test-support"))] + { + self.fault.swap(false, Ordering::SeqCst) + } + #[cfg(not(any(test, feature = "test-support")))] + { + false + } + } + /// One pass over every Petri run the database holds: the runs with a /// Petri record, and the runs with platform records. Runs whose view /// already covers every committed record are skipped cheaply. @@ -429,67 +492,20 @@ impl Projector { /// positions the cache's view holds, fold it, and write the view. async fn pass(&self, run_id: RunId, run: &mut RunCache) -> Result { let key = RunKey::new(run_id.to_string()); - let platform_head = self - .platform - .head(&run_id) - .await - .map_err(ProjectError::Store)? - .unwrap_or(0); - let petri_heads = self.petri_heads(&run_id).await?; - let stored = &run.view; - let at_head = platform_head == stored.positions.platform_seq - && petri_heads.iter().all(|(log, head)| { - stored - .positions - .petri - .iter() - .any(|held| log_text(&held.source) == *log && held.seq == *head) - }); - if at_head && stored.view.projection.is_some() { - return Ok(PassReport { - run_id, - skipped: true, - contended: false, - petri_events: 0, - platform_records: 0, - replayed_records: 0, - stream_seq: stored.stream_seq, - positions: stored.positions.clone(), - health: stored.view.state.health.clone(), - }); + if self.at_head(run_id, &run.view).await? { + return Ok(PassReport::skipped(run_id, &run.view)); } let platform_records = self .platform - .read_after(&run_id, stored.positions.platform_seq) + .read_after(&run_id, run.view.positions.platform_seq) .await .map_err(ProjectError::Store)?; - let mut replayed_records = 0; - let (events, replay_failure) = match self.store.open(&key, Access::Read).await { - Ok(logs) => match run.replay.advance(&*logs).await { - Ok(new) => { - replayed_records = new.iter().filter(|event| event.id.index == 0).count(); - // A rebuilt replay derives the run whole: only the events - // past the view's positions are new to it. - let held = run.view.positions.held(); - let mut events = std::mem::take(&mut run.pending); - events.extend(new.into_iter().filter(|event| { - held.get(&event.id.source) - .is_none_or(|last| event.id > *last) - })); - (events, None) - } - // The replay stood still and is retried by the next pass; - // what it derived before stays pending. - Err(error) => { - let chain = collect_chain(&error).join(": "); - warn!(run_id = %run_id, error = %chain, "Petri run does not replay; the view holds"); - (Vec::new(), Some(chain)) - } - }, - Err(StoreError::NotFound { .. }) => (Vec::new(), None), - Err(error) => return Err(ProjectError::Open(error)), - }; + let NewEvents { + events, + replay_failure, + replayed_records, + } = self.read_new_events(run_id, &key, run).await?; let mut view = run.view.view.clone(); let mut positions = run.view.positions.clone(); @@ -504,7 +520,7 @@ impl Projector { let platform_head_seen = platform_records .last() .map_or(positions.platform_seq, |record| record.seq); - let (items, held) = order_items( + let (items, held) = order::order_items( &events, &platform_records, &view.state.finished_firings, @@ -517,37 +533,11 @@ impl Projector { "platform records held back until their firing's finish is in the stream" ); } - - let mut rows: Vec = Vec::with_capacity(items.len()); - for item in &items { - stream_seq += 1; - view.fold(item, stream_seq); - let row = match item { - Item::Petri(event) => { - positions.advance(event.id); - StreamRow { - stream_seq, - item_kind: "petri", - item_id: event_id_text(&event.id), - event_json: serde_json::to_string(event).map_err(ProjectError::Encode)?, - } - } - Item::Platform(record) => { - positions.platform_seq = record.seq; - StreamRow { - stream_seq, - item_kind: "platform", - item_id: record.seq.to_string(), - event_json: serde_json::to_string(record).map_err(ProjectError::Encode)?, - } - } - }; - rows.push(row); - } + let rows = stream::stream_rows(&items, &mut view, &mut positions, &mut stream_seq)?; drop(items); view.state.health = self.health(&key, &view.state, replay_failure).await?; - if self.fault.swap(false, Ordering::SeqCst) { + if self.take_fault() { run.pending = events; return Err(ProjectError::Injected); } @@ -566,17 +556,7 @@ impl Projector { Ok(false) => { debug!(run_id = %run_id, "platform records landed during the pass; running it again"); run.pending = events; - return Ok(PassReport { - run_id, - skipped: false, - contended: true, - petri_events: 0, - platform_records: 0, - replayed_records, - stream_seq: 0, - positions: Positions::default(), - health: RecordHealth::default(), - }); + return Ok(PassReport::contended(run_id, replayed_records)); } Err(error) => { run.pending = events; @@ -611,6 +591,77 @@ impl Projector { }) } + /// Whether the stored view already covers every committed record: the + /// platform head and every Petri log's head are the positions it holds, + /// and it has a projection to serve. + async fn at_head(&self, run_id: RunId, stored: &StoredView) -> Result { + let platform_head = self + .platform + .head(&run_id) + .await + .map_err(ProjectError::Store)? + .unwrap_or(0); + let petri_heads = self.petri_heads(&run_id).await?; + Ok(platform_head == stored.positions.platform_seq + && petri_heads.iter().all(|(log, head)| { + stored + .positions + .petri + .iter() + .any(|held| stream::log_text(&held.source) == *log && held.seq == *head) + }) + && stored.view.projection.is_some()) + } + + /// The events past the view's positions: the cache's replay advanced + /// over the records committed since, led by the events an earlier pass + /// derived and did not commit. A rebuilt replay derives the run whole, + /// so only the events past the view's positions are new to it. A + /// replay that fails (a torn tail) stands still, yields nothing, and + /// names its reason; what it derived before stays pending for the next + /// pass. + async fn read_new_events( + &self, + run_id: RunId, + key: &RunKey, + run: &mut RunCache, + ) -> Result { + let nothing = NewEvents { + events: Vec::new(), + replay_failure: None, + replayed_records: 0, + }; + let logs = match self.store.open(key, Access::Read).await { + Ok(logs) => logs, + Err(StoreError::NotFound { .. }) => return Ok(nothing), + Err(error) => return Err(ProjectError::Open(error)), + }; + match run.replay.advance(&*logs).await { + Ok(new) => { + let replayed_records = new.iter().filter(|event| event.id.index == 0).count(); + let held = run.view.positions.held(); + let mut events = std::mem::take(&mut run.pending); + events.extend(new.into_iter().filter(|event| { + held.get(&event.id.source) + .is_none_or(|last| event.id > *last) + })); + Ok(NewEvents { + events, + replay_failure: None, + replayed_records, + }) + } + Err(error) => { + let chain = collect_chain(&error).join(": "); + warn!(run_id = %run_id, error = %chain, "Petri run does not replay; the view holds"); + Ok(NewEvents { + replay_failure: Some(chain), + ..nothing + }) + } + } + } + /// The view transaction: the projection row, the stream rows and the /// `runs` row, committed together, unless a platform record landed /// since the pass read them (`false`: the view is left alone). @@ -654,8 +705,8 @@ impl Projector { .bind(projection_json) .bind(fold_json) .bind(positions_json) - .bind(column(stream_seq)) - .bind(column(now_ms())) + .bind(stream::column(stream_seq)) + .bind(stream::column(now_ms())) .execute(&mut *tx) .await .map_err(ProjectError::Database)?; @@ -665,7 +716,7 @@ impl Projector { VALUES (?, ?, ?, ?, ?)", ) .bind(run_id.to_string()) - .bind(column(row.stream_seq)) + .bind(stream::column(row.stream_seq)) .bind(row.item_kind) .bind(&row.item_id) .bind(&row.event_json) @@ -771,345 +822,13 @@ impl Projector { } } -impl Projector { - /// A run store whose appends signal this projector: for a run that - /// executes in the same process as the projector, over the SQLite store - /// directly, where no append endpoint is there to signal. The signal is - /// sent after the store's append returned, so the records it covers are - /// durable before the view sees them. - pub fn observe_store( - self: &Arc, - inner: Arc, - ) -> Arc { - Arc::new(SignallingStore { - inner, - projector: Arc::clone(self), - }) - } -} - -/// A run store that signals a projector after each append. -struct SignallingStore { - inner: Arc, - projector: Arc, -} - -#[async_trait::async_trait] -impl petri_execution::RunStore for SignallingStore { - async fn open( - &self, - key: &RunKey, - access: Access, - ) -> Result, StoreError> { - let logs = self.inner.open(key, access).await?; - Ok(Arc::new(SignallingLogs { - inner: logs, - run_id: projection::run_id_of(key.as_str()), - projector: Arc::clone(&self.projector), - })) - } -} - -struct SignallingLogs { - inner: Arc, - run_id: Option, - projector: Arc, -} - -#[async_trait::async_trait] -impl petri_execution::RunLogs for SignallingLogs { - fn locator(&self) -> String { - self.inner.locator() - } - - async fn append( - &self, - log: &petri_execution::LogId, - records: &[petri_execution::Record], - ) -> Result<(), StoreError> { - self.inner.append(log, records).await?; - if let Some(run_id) = self.run_id { - self.projector.signal(run_id); - } - Ok(()) - } - - async fn read( - &self, - log: &petri_execution::LogId, - ) -> Result, StoreError> { - self.inner.read(log).await - } - - async fn read_from( - &self, - log: &petri_execution::LogId, - seq: u64, - ) -> Result, StoreError> { - self.inner.read_from(log, seq).await - } - - async fn put_blob(&self, bytes: &[u8]) -> Result { - self.inner.put_blob(bytes).await - } - - async fn get_blob(&self, digest: petri_store::Digest) -> Result>, StoreError> { - self.inner.get_blob(digest).await - } -} - -/// The order one pass streams its new items in, and how many platform -/// records it holds back for a later pass. -/// -/// Every item is first ordered by `recorded_at` (stable: the coordinator -/// log before an execution log before a platform record on a tie, and each -/// log's own order kept). A platform record that carries a Petri position -/// (a checkpoint, keyed on `(execution, firing)`) is then placed by that -/// position, not by its clock, because the server stamps the record and the -/// worker stamps Petri's records and the two clocks can tie or invert: -/// -/// - before the firing's first `routing.resolved` event in the pass, which is -/// right after the firing's finish (its `step.finished` and the -/// `visit.completed` attached to it) and before the next firing's -/// `visit.started`, which is attached to that routing record; -/// - else after the last event of the firing in the pass; -/// - else, when the firing finished in an earlier pass, before the first event -/// of a later firing (a larger firing id) in the same execution, or where its -/// `recorded_at` put it; -/// - else the record is held back, with every platform record after it, and the -/// pass consumes platform records only up to it. The hook that writes a -/// checkpoint record runs after the driver appended the attempt's finish, but -/// the driver's store writer flushes that record on its own schedule, so the -/// platform record can be committed before its firing's `step.finished`; -/// holding it keeps the stream's order the same live and on a rebuild. -/// Nothing is held once the run has recorded its finish. -/// -/// The rule reads only the pass's own items and the firings already -/// finished, so a record is never streamed before its firing's finish and -/// never after the firing's routes. -fn order_items<'a>( - events: &'a [RunEvent], - platform_records: &'a [StoredPlatformRecord], - finished_before: &BTreeSet, - run_finished: bool, -) -> (Vec>, usize) { - let firing_of = |event: &RunEvent| -> Option<(u64, u64)> { - let execution = event.context.execution?; - let firing = event.subject.as_ref()?.firing?; - Some((execution.raw(), firing.raw())) - }; - let finished_in_pass = |at: (u64, u64)| { - events.iter().any(|event| { - firing_of(event) == Some(at) - && matches!(event.engine(), Some(Event::StepFinished { .. })) - }) - }; - let finished = |at: (u64, u64)| { - finished_in_pass(at) || finished_before.contains(&projection::stage_key(at.0, at.1)) - }; - // Platform records are consumed in seq order: the first one whose firing - // has not finished holds itself and everything after it. - let consumed = if run_finished { - platform_records.len() - } else { - platform_records - .iter() - .position(|record| { - record - .position - .is_some_and(|position| !finished((position.execution, position.firing))) - }) - .unwrap_or(platform_records.len()) - }; - let held = platform_records.len() - consumed; - let platform_records = &platform_records[..consumed]; - - let mut items: Vec<(u64, u8, Item<'a>)> = - Vec::with_capacity(events.len() + platform_records.len()); - for event in events { - let rank = match event.id.source { - EventSource::Coordinator => 0, - EventSource::Execution { .. } => 1, - }; - items.push((event.recorded_at, rank, Item::Petri(event))); - } - for record in platform_records { - items.push((record.recorded_at, 2, Item::Platform(record))); - } - items.sort_by_key(|(recorded_at, rank, _)| (*recorded_at, *rank)); - - let item_firing = |item: &Item<'a>| match item { - Item::Petri(event) => firing_of(event), - Item::Platform(_) => None, - }; - let is_routing = |item: &Item<'a>| { - matches!( - item, - Item::Petri(event) if matches!(event.engine(), Some(Event::RoutingResolved { .. })) - ) - }; - // The key of each item: its index in clock order, and whether it sits - // before (0), at (1) or after (2) that index. - let mut keys: Vec<(usize, u8)> = (0..items.len()).map(|index| (index, 1)).collect(); - for (index, (_, _, item)) in items.iter().enumerate() { - let Item::Platform(record) = item else { - continue; - }; - let Some(position) = record.position else { - continue; - }; - let at = (position.execution, position.firing); - let first_routing = items - .iter() - .position(|(_, _, other)| item_firing(other) == Some(at) && is_routing(other)); - let last_of_firing = items - .iter() - .rposition(|(_, _, other)| item_firing(other) == Some(at)); - let first_later = items.iter().position(|(_, _, other)| { - item_firing(other).is_some_and(|(execution, firing)| execution == at.0 && firing > at.1) - }); - keys[index] = if let Some(before) = first_routing { - (before, 0) - } else if let Some(after) = last_of_firing { - (after, 2) - } else if let Some(before) = first_later { - (before, 0) - } else { - (index, 1) - }; - } - let mut order: Vec = (0..items.len()).collect(); - order.sort_by_key(|index| keys[*index]); - let mut ordered: Vec>> = - items.into_iter().map(|(_, _, item)| Some(item)).collect(); - let items = order - .into_iter() - .map(|index| ordered[index].take().expect("each item is placed once")) - .collect(); - (items, held) -} - -struct StreamRow { - stream_seq: u64, - item_kind: &'static str, - item_id: String, - event_json: String, -} - /// How many commit signals a slow reader may fall behind before it is told /// it lagged and re-reads from its cursor. const COMMIT_SIGNAL_CAPACITY: usize = 1024; -/// The run's stream past the cursor, read from the view tables: up to -/// `limit` rows with `stream_seq > after`, in order, in Fabro's envelope. -pub async fn stream_after( - views: &DbPool, - run_id: RunId, - after: u64, - limit: usize, -) -> Result, ProjectError> { - let rows: Vec<(i64, String, String, String)> = sqlx::query_as( - "SELECT stream_seq, item_kind, item_id, event_json FROM petri_stream WHERE run_id = ? AND \ - stream_seq > ? ORDER BY stream_seq LIMIT ?", - ) - .bind(run_id.to_string()) - .bind(column(after)) - .bind(i64::try_from(limit).unwrap_or(i64::MAX)) - .fetch_all(views) - .await - .map_err(ProjectError::Database)?; - rows.into_iter() - .map(|(stream_seq, item_kind, item_id, event_json)| { - let item: serde_json::Value = - serde_json::from_str(&event_json).map_err(ProjectError::Encode)?; - let kind = match item_kind.as_str() { - "platform" => RunStreamItemKind::Platform, - _ => RunStreamItemKind::Petri, - }; - let recorded_at = item - .get("recorded_at") - .and_then(serde_json::Value::as_u64) - .unwrap_or(0); - Ok(RunStreamItem { - run_id, - stream_seq: u64::try_from(stream_seq).unwrap_or(0), - kind, - id: item_id, - recorded_at, - item, - }) - }) - .collect() -} - -/// A Petri event id as the stream names it: `//`. -#[must_use] -pub fn event_id_text(id: &EventId) -> String { - format!("{}/{}/{}", log_text(&id.source), id.seq, id.index) -} - -fn log_text(source: &EventSource) -> String { - match source { - EventSource::Coordinator => "coordinator".to_string(), - EventSource::Execution { execution } => format!("execution {execution}"), - } -} - -fn column(value: u64) -> i64 { - i64::try_from(value).unwrap_or(i64::MAX) -} - -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - -/// The run's projection rebuilt from its records alone, with nothing -/// stored: what a fresh projector would commit over the same records. A test -/// compares it with the live view. `records` and `views` are the two pools -/// [`Projector::new`] takes. -pub async fn rebuild( - records: &DbPool, - views: &DbPool, - run_id: RunId, -) -> Result<(Option, Positions, u64), ProjectError> { - let store = SqliteRunStore::new(records.clone()); - let platform = PlatformRecordStore::new(views.clone()); - let key = RunKey::new(run_id.to_string()); - let platform_records = platform.read(&run_id).await.map_err(ProjectError::Store)?; - let events = match store.open(&key, Access::Read).await { - Ok(logs) => events::replay_run(&*logs) - .await - .inspect_err(|error| { - warn!(error = %collect_chain(error).join(": "), "rebuild: the run does not replay"); - }) - .unwrap_or_default(), - Err(StoreError::NotFound { .. }) => Vec::new(), - Err(error) => return Err(ProjectError::Open(error)), - }; - let run_finished = events.iter().any(|event| { - matches!( - event.coordinator(), - Some(CoordinatorEvent::RunFinished { .. }) - ) - }); - let (items, _held) = order_items(&events, &platform_records, &BTreeSet::new(), run_finished); - let mut view = RunView::new(); - let mut positions = Positions::default(); - let mut stream_seq = 0; - for item in &items { - stream_seq += 1; - view.fold(item, stream_seq); - match item { - Item::Petri(event) => positions.advance(event.id), - Item::Platform(record) => positions.platform_seq = record.seq, - } - } - Ok((view.projection, positions, stream_seq)) -} - -/// The stored view's positions and stream sequence, for a test; `views` is -/// the pool the view tables live in. -pub async fn stored_positions( +/// The stored view's positions and stream sequence; `views` is the pool the +/// view tables live in. +pub(crate) async fn stored_positions( views: &DbPool, run_id: RunId, ) -> Result, ProjectError> { @@ -1127,261 +846,3 @@ pub async fn stored_positions( }) .transpose() } - -/// The stored view's projection, for a test or a reader outside the store. -pub async fn stored_projection( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - let json: Option = - sqlx::query_scalar("SELECT projection_json FROM petri_projection WHERE run_id = ?") - .bind(run_id.to_string()) - .fetch_optional(views) - .await - .map_err(ProjectError::Database)?; - json.map(|json| serde_json::from_str(&json).map_err(ProjectError::Encode)) - .transpose() -} - -/// The stream rows of a run: `(stream_seq, item_kind, item_id)`, in order. -pub async fn stored_stream( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - let rows: Vec<(i64, String, String)> = sqlx::query_as( - "SELECT stream_seq, item_kind, item_id FROM petri_stream WHERE run_id = ? ORDER BY stream_seq", - ) - .bind(run_id.to_string()) - .fetch_all(views) - .await - .map_err(ProjectError::Database)?; - Ok(rows - .into_iter() - .map(|(seq, kind, id)| (u64::try_from(seq).unwrap_or(0), kind, id)) - .collect()) -} - -/// Every stored platform record of a run, for a reader outside the store. -pub async fn stored_platform_records( - views: &DbPool, - run_id: RunId, -) -> Result, ProjectError> { - PlatformRecordStore::new(views.clone()) - .read(&run_id) - .await - .map_err(ProjectError::Store) -} - -/// A recorded event's projection is what `RunEvent` serializes to. -#[must_use] -pub fn event_json(event: &RunEvent) -> serde_json::Value { - serde_json::to_value(event).unwrap_or_default() -} - -#[cfg(test)] -mod tests { - use fabro_store::PlatformRecord; - use fabro_store::platform_records::{CheckpointRecord, StagePosition}; - use petri_execution::events::{Context, NodeRef, Record, RecordOrigin, Subject}; - use petri_execution::{ExecutionId, StoredEngineRecord}; - use petri_runtime::driver::BranchRole; - use petri_runtime::engine::{DecisionId, EventOrigin, RouteApplied}; - use petri_runtime::ir::{Attempt, FiringId, NodeId, Outcome, Status}; - - use super::*; - - /// A firing's engine event at `seq`, recorded at `at`. - fn engine_event(seq: u64, firing: u64, at: u64, body: Event) -> RunEvent { - RunEvent { - id: EventId { - source: EventSource::Execution { - execution: ExecutionId::new(0), - }, - seq, - index: 0, - }, - origin: RecordOrigin::External, - context: Context { - invocation: None, - execution: Some(ExecutionId::new(0)), - parent: None, - }, - subject: Some(Subject { - node: NodeRef { - id: NodeId::new(1), - name: format!("n{firing}").into(), - kind: "attractor/command".into(), - meta: serde_json::Value::Null, - }, - firing: Some(FiringId::new(firing)), - visit: Some(1), - attempt: Some(Attempt::FIRST), - generation: None, - branch: BranchRole::None, - }), - observed_at: None, - recorded_at: at, - record: Some(Record::Engine(StoredEngineRecord { - seq, - origin: EventOrigin::External, - recorded_at: at, - body, - })), - derived: None, - } - } - - fn finished(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::StepFinished { - firing: FiringId::new(firing), - attempt: Attempt::FIRST, - outcome: Outcome::new(Status::Success, serde_json::Value::Null), - }) - } - - fn routing(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::RoutingResolved { - decision_id: DecisionId::route(FiringId::new(firing), Attempt::FIRST), - groups: Vec::new(), - }) - } - - fn applied(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::RouteApplied { - applied: RouteApplied::None { - firing: FiringId::new(firing), - group: 0, - }, - }) - } - - fn started(seq: u64, firing: u64, at: u64) -> RunEvent { - engine_event(seq, firing, at, Event::StepStarted { - firing: FiringId::new(firing), - attempt: Attempt::FIRST, - }) - } - - fn checkpoint(seq: u64, firing: u64, at: u64) -> StoredPlatformRecord { - StoredPlatformRecord { - seq, - recorded_at: at, - record: PlatformRecord::Checkpoint(CheckpointRecord { - execution: 0, - firing, - attempt: Some(1), - workspace: None, - git_commit_sha: Some("abc".to_string()), - diff_summary: None, - patch_blob: None, - operation: None, - }), - position: Some(StagePosition { - execution: 0, - firing, - }), - } - } - - fn names(items: &[Item<'_>]) -> Vec { - items - .iter() - .map(|item| match item { - Item::Petri(event) => event_id_text(&event.id), - Item::Platform(record) => format!("platform {}", record.seq), - }) - .collect() - } - - /// A firing's events, then a later firing's events, then a checkpoint - /// for the first firing stamped later than all of them: the stream puts - /// the checkpoint right after the first firing's finish, before its - /// routes and before the later firing. - #[test] - fn a_positioned_record_follows_its_firings_finish_whatever_its_clock_says() { - let events = vec![ - finished(10, 1, 100), - routing(11, 1, 101), - applied(12, 1, 102), - started(13, 2, 103), - finished(14, 2, 104), - ]; - let records = vec![checkpoint(1, 1, 250)]; - let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "execution 0/10/0", - "platform 1", - "execution 0/11/0", - "execution 0/12/0", - "execution 0/13/0", - "execution 0/14/0", - ]); - } - - /// With the firing finished in an earlier pass, the record goes before - /// the first event of a later firing; a record with no position keeps - /// its clock order. - #[test] - fn a_positioned_record_precedes_later_firings_and_an_unpositioned_one_keeps_its_clock() { - let events = vec![started(13, 2, 103), finished(14, 2, 104)]; - let records = vec![checkpoint(1, 1, 250)]; - let finished_before: BTreeSet = [projection::stage_key(0, 1)].into_iter().collect(); - let (items, held) = order_items(&events, &records, &finished_before, false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "platform 1", - "execution 0/13/0", - "execution 0/14/0", - ]); - - let unpositioned = StoredPlatformRecord { - position: None, - ..checkpoint(2, 1, 250) - }; - let unpositioned = [unpositioned]; - let (items, held) = order_items(&events, &unpositioned, &BTreeSet::new(), false); - assert_eq!(held, 0); - assert_eq!(names(&items), vec![ - "execution 0/13/0", - "execution 0/14/0", - "platform 2", - ]); - } - - /// A record whose firing has no finish yet, in the stream or in the pass, - /// is held back with everything after it until the finish arrives, or - /// until the run has finished. - #[test] - fn a_positioned_record_is_held_until_its_firings_finish_is_in_the_stream() { - let events = vec![started(13, 2, 103), finished(14, 2, 104)]; - let records = vec![ - checkpoint(1, 2, 50), - checkpoint(2, 3, 60), - checkpoint(3, 2, 70), - ]; - let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); - assert_eq!( - held, 2, - "the record for firing 3 holds itself and the one after it" - ); - assert_eq!(names(&items), vec![ - "execution 0/13/0", - "execution 0/14/0", - "platform 1", - ]); - - // Once the run finished, a record for a firing that never finished - // keeps its clock order; the firing's own records still follow its - // finish, in their seq order. - let (items, held) = order_items(&events, &records, &BTreeSet::new(), true); - assert_eq!(held, 0, "nothing is held once the run finished"); - assert_eq!(names(&items), vec![ - "platform 2", - "execution 0/13/0", - "execution 0/14/0", - "platform 1", - "platform 3", - ]); - } -} diff --git a/lib/components/fabro-petri/src/projector/order.rs b/lib/components/fabro-petri/src/projector/order.rs new file mode 100644 index 000000000..2b27aaff4 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/order.rs @@ -0,0 +1,343 @@ +//! The order one pass streams its new items in. + +use std::collections::BTreeSet; + +use fabro_store::platform_records::StoredPlatformRecord; +use petri_execution::events::{EventSource, RunEvent}; +use petri_runtime::engine::Event; + +use crate::projection::{FiringKey, Item}; + +/// The order one pass streams its new items in, and how many platform +/// records it holds back for a later pass. +/// +/// Every item is first ordered by `recorded_at` (stable: the coordinator +/// log before an execution log before a platform record on a tie, and each +/// log's own order kept). A platform record that carries a Petri position +/// (a checkpoint, keyed on `(execution, firing)`) is then placed by that +/// position, not by its clock, because the server stamps the record and the +/// worker stamps Petri's records and the two clocks can tie or invert: +/// +/// - before the firing's first `routing.resolved` event in the pass, which is +/// right after the firing's finish (its `step.finished` and the +/// `visit.completed` attached to it) and before the next firing's +/// `visit.started`, which is attached to that routing record; +/// - else after the last event of the firing in the pass; +/// - else, when the firing finished in an earlier pass, before the first event +/// of a later firing (a larger firing id) in the same execution, or where its +/// `recorded_at` put it; +/// - else the record is held back, with every platform record after it, and the +/// pass consumes platform records only up to it. The hook that writes a +/// checkpoint record runs after the driver appended the attempt's finish, but +/// the driver's store writer flushes that record on its own schedule, so the +/// platform record can be committed before its firing's `step.finished`; +/// holding it keeps the stream's order the same live and on a rebuild. +/// Nothing is held once the run has recorded its finish. +/// +/// The rule reads only the pass's own items and the firings already +/// finished, so a record is never streamed before its firing's finish and +/// never after the firing's routes. +pub(crate) fn order_items<'a>( + events: &'a [RunEvent], + platform_records: &'a [StoredPlatformRecord], + finished_before: &BTreeSet, + run_finished: bool, +) -> (Vec>, usize) { + let finished_in_pass = |at: FiringKey| { + events.iter().any(|event| { + FiringKey::of_event(event) == Some(at) + && matches!(event.engine(), Some(Event::StepFinished { .. })) + }) + }; + let finished = |at: FiringKey| finished_in_pass(at) || finished_before.contains(&at); + // Platform records are consumed in seq order: the first one whose firing + // has not finished holds itself and everything after it. + let consumed = if run_finished { + platform_records.len() + } else { + platform_records + .iter() + .position(|record| { + record + .position + .is_some_and(|position| !finished(FiringKey::from(position))) + }) + .unwrap_or(platform_records.len()) + }; + let held = platform_records.len() - consumed; + let platform_records = &platform_records[..consumed]; + + let mut items: Vec<(u64, u8, Item<'a>)> = + Vec::with_capacity(events.len() + platform_records.len()); + for event in events { + let rank = match event.id.source { + EventSource::Coordinator => 0, + EventSource::Execution { .. } => 1, + }; + items.push((event.recorded_at, rank, Item::Petri(event))); + } + for record in platform_records { + items.push((record.recorded_at, 2, Item::Platform(record))); + } + items.sort_by_key(|(recorded_at, rank, _)| (*recorded_at, *rank)); + + let item_firing = |item: &Item<'a>| match item { + Item::Petri(event) => FiringKey::of_event(event), + Item::Platform(_) => None, + }; + let is_routing = |item: &Item<'a>| { + matches!( + item, + Item::Petri(event) if matches!(event.engine(), Some(Event::RoutingResolved { .. })) + ) + }; + // The key of each item: its index in clock order, and whether it sits + // before (0), at (1) or after (2) that index. + let mut keys: Vec<(usize, u8)> = (0..items.len()).map(|index| (index, 1)).collect(); + for (index, (_, _, item)) in items.iter().enumerate() { + let Item::Platform(record) = item else { + continue; + }; + let Some(position) = record.position else { + continue; + }; + let at = FiringKey::from(position); + let first_routing = items + .iter() + .position(|(_, _, other)| item_firing(other) == Some(at) && is_routing(other)); + let last_of_firing = items + .iter() + .rposition(|(_, _, other)| item_firing(other) == Some(at)); + let first_later = items.iter().position(|(_, _, other)| { + item_firing(other) + .is_some_and(|key| key.execution == at.execution && key.firing > at.firing) + }); + keys[index] = if let Some(before) = first_routing { + (before, 0) + } else if let Some(after) = last_of_firing { + (after, 2) + } else if let Some(before) = first_later { + (before, 0) + } else { + (index, 1) + }; + } + let mut order: Vec = (0..items.len()).collect(); + order.sort_by_key(|index| keys[*index]); + let mut ordered: Vec>> = + items.into_iter().map(|(_, _, item)| Some(item)).collect(); + let items = order + .into_iter() + .map(|index| ordered[index].take().expect("each item is placed once")) + .collect(); + (items, held) +} + +#[cfg(test)] +mod tests { + use fabro_store::PlatformRecord; + use fabro_store::platform_records::{CheckpointRecord, StagePosition}; + use petri_execution::events::{Context, EventId, NodeRef, Record, RecordOrigin, Subject}; + use petri_execution::{ExecutionId, StoredEngineRecord}; + use petri_runtime::driver::BranchRole; + use petri_runtime::engine::{DecisionId, EventOrigin, RouteApplied}; + use petri_runtime::ir::{Attempt, FiringId, NodeId, Outcome, Status}; + + use super::*; + use crate::projector::stream; + + /// A firing's engine event at `seq`, recorded at `at`. + fn engine_event(seq: u64, firing: u64, at: u64, body: Event) -> RunEvent { + RunEvent { + id: EventId { + source: EventSource::Execution { + execution: ExecutionId::new(0), + }, + seq, + index: 0, + }, + origin: RecordOrigin::External, + context: Context { + invocation: None, + execution: Some(ExecutionId::new(0)), + parent: None, + }, + subject: Some(Subject { + node: NodeRef { + id: NodeId::new(1), + name: format!("n{firing}").into(), + kind: "attractor/command".into(), + meta: serde_json::Value::Null, + }, + firing: Some(FiringId::new(firing)), + visit: Some(1), + attempt: Some(Attempt::FIRST), + generation: None, + branch: BranchRole::None, + }), + observed_at: None, + recorded_at: at, + record: Some(Record::Engine(StoredEngineRecord { + seq, + origin: EventOrigin::External, + recorded_at: at, + body, + })), + derived: None, + } + } + + fn finished(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::StepFinished { + firing: FiringId::new(firing), + attempt: Attempt::FIRST, + outcome: Outcome::new(Status::Success, serde_json::Value::Null), + }) + } + + fn routing(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::RoutingResolved { + decision_id: DecisionId::route(FiringId::new(firing), Attempt::FIRST), + groups: Vec::new(), + }) + } + + fn applied(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::RouteApplied { + applied: RouteApplied::None { + firing: FiringId::new(firing), + group: 0, + }, + }) + } + + fn started(seq: u64, firing: u64, at: u64) -> RunEvent { + engine_event(seq, firing, at, Event::StepStarted { + firing: FiringId::new(firing), + attempt: Attempt::FIRST, + }) + } + + fn checkpoint(seq: u64, firing: u64, at: u64) -> StoredPlatformRecord { + StoredPlatformRecord { + seq, + recorded_at: at, + record: PlatformRecord::Checkpoint(CheckpointRecord { + execution: 0, + firing, + attempt: Some(1), + workspace: None, + git_commit_sha: Some("abc".to_string()), + diff_summary: None, + patch_blob: None, + operation: None, + }), + position: Some(StagePosition { + execution: 0, + firing, + }), + } + } + + fn names(items: &[Item<'_>]) -> Vec { + items + .iter() + .map(|item| match item { + Item::Petri(event) => stream::event_id_text(&event.id), + Item::Platform(record) => format!("platform {}", record.seq), + }) + .collect() + } + + /// A firing's events, then a later firing's events, then a checkpoint + /// for the first firing stamped later than all of them: the stream puts + /// the checkpoint right after the first firing's finish, before its + /// routes and before the later firing. + #[test] + fn a_positioned_record_follows_its_firings_finish_whatever_its_clock_says() { + let events = vec![ + finished(10, 1, 100), + routing(11, 1, 101), + applied(12, 1, 102), + started(13, 2, 103), + finished(14, 2, 104), + ]; + let records = vec![checkpoint(1, 1, 250)]; + let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "execution 0/10/0", + "platform 1", + "execution 0/11/0", + "execution 0/12/0", + "execution 0/13/0", + "execution 0/14/0", + ]); + } + + /// With the firing finished in an earlier pass, the record goes before + /// the first event of a later firing; a record with no position keeps + /// its clock order. + #[test] + fn a_positioned_record_precedes_later_firings_and_an_unpositioned_one_keeps_its_clock() { + let events = vec![started(13, 2, 103), finished(14, 2, 104)]; + let records = vec![checkpoint(1, 1, 250)]; + let finished_before: BTreeSet = [FiringKey::new(0, 1)].into_iter().collect(); + let (items, held) = order_items(&events, &records, &finished_before, false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "platform 1", + "execution 0/13/0", + "execution 0/14/0", + ]); + + let unpositioned = StoredPlatformRecord { + position: None, + ..checkpoint(2, 1, 250) + }; + let unpositioned = [unpositioned]; + let (items, held) = order_items(&events, &unpositioned, &BTreeSet::new(), false); + assert_eq!(held, 0); + assert_eq!(names(&items), vec![ + "execution 0/13/0", + "execution 0/14/0", + "platform 2", + ]); + } + + /// A record whose firing has no finish yet, in the stream or in the pass, + /// is held back with everything after it until the finish arrives, or + /// until the run has finished. + #[test] + fn a_positioned_record_is_held_until_its_firings_finish_is_in_the_stream() { + let events = vec![started(13, 2, 103), finished(14, 2, 104)]; + let records = vec![ + checkpoint(1, 2, 50), + checkpoint(2, 3, 60), + checkpoint(3, 2, 70), + ]; + let (items, held) = order_items(&events, &records, &BTreeSet::new(), false); + assert_eq!( + held, 2, + "the record for firing 3 holds itself and the one after it" + ); + assert_eq!(names(&items), vec![ + "execution 0/13/0", + "execution 0/14/0", + "platform 1", + ]); + + // Once the run finished, a record for a firing that never finished + // keeps its clock order; the firing's own records still follow its + // finish, in their seq order. + let (items, held) = order_items(&events, &records, &BTreeSet::new(), true); + assert_eq!(held, 0, "nothing is held once the run finished"); + assert_eq!(names(&items), vec![ + "platform 2", + "execution 0/13/0", + "execution 0/14/0", + "platform 1", + "platform 3", + ]); + } +} diff --git a/lib/components/fabro-petri/src/projector/signalling.rs b/lib/components/fabro-petri/src/projector/signalling.rs new file mode 100644 index 000000000..c81a97587 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/signalling.rs @@ -0,0 +1,99 @@ +//! The in-process wake-up: a run store whose appends signal the projector, +//! for a run that executes in the same process as the projector, over the +//! SQLite store directly, where no append endpoint is there to signal. + +use std::sync::Arc; + +use fabro_types::RunId; +use petri_execution::{Access, RunKey}; +use petri_store::StoreError; + +use super::Projector; +use crate::projection; + +impl Projector { + /// A run store whose appends signal this projector: for a run that + /// executes in the same process as the projector, over the SQLite store + /// directly, where no append endpoint is there to signal. The signal is + /// sent after the store's append returned, so the records it covers are + /// durable before the view sees them. + pub fn observe_store( + self: &Arc, + inner: Arc, + ) -> Arc { + Arc::new(SignallingStore { + inner, + projector: Arc::clone(self), + }) + } +} + +/// A run store that signals a projector after each append. +struct SignallingStore { + inner: Arc, + projector: Arc, +} + +#[async_trait::async_trait] +impl petri_execution::RunStore for SignallingStore { + async fn open( + &self, + key: &RunKey, + access: Access, + ) -> Result, StoreError> { + let logs = self.inner.open(key, access).await?; + Ok(Arc::new(SignallingLogs { + inner: logs, + run_id: projection::run_id_of(key.as_str()), + projector: Arc::clone(&self.projector), + })) + } +} + +struct SignallingLogs { + inner: Arc, + run_id: Option, + projector: Arc, +} + +#[async_trait::async_trait] +impl petri_execution::RunLogs for SignallingLogs { + fn locator(&self) -> String { + self.inner.locator() + } + + async fn append( + &self, + log: &petri_execution::LogId, + records: &[petri_execution::Record], + ) -> Result<(), StoreError> { + self.inner.append(log, records).await?; + if let Some(run_id) = self.run_id { + self.projector.signal(run_id); + } + Ok(()) + } + + async fn read( + &self, + log: &petri_execution::LogId, + ) -> Result, StoreError> { + self.inner.read(log).await + } + + async fn read_from( + &self, + log: &petri_execution::LogId, + seq: u64, + ) -> Result, StoreError> { + self.inner.read_from(log, seq).await + } + + async fn put_blob(&self, bytes: &[u8]) -> Result { + self.inner.put_blob(bytes).await + } + + async fn get_blob(&self, digest: petri_store::Digest) -> Result>, StoreError> { + self.inner.get_blob(digest).await + } +} diff --git a/lib/components/fabro-petri/src/projector/stream.rs b/lib/components/fabro-petri/src/projector/stream.rs new file mode 100644 index 000000000..f35e63d10 --- /dev/null +++ b/lib/components/fabro-petri/src/projector/stream.rs @@ -0,0 +1,113 @@ +//! The run's stream as the view tables hold it: one row per folded item, +//! written by a pass and read back in `stream_seq` order. + +use fabro_db::DbPool; +use fabro_types::{RunId, RunStreamItem, RunStreamItemKind}; +use petri_execution::events::{EventId, EventSource}; + +use super::{Positions, ProjectError}; +use crate::projection::{Item, RunView}; + +pub(crate) struct StreamRow { + pub(super) stream_seq: u64, + pub(super) item_kind: &'static str, + pub(super) item_id: String, + pub(super) event_json: String, +} + +/// Fold the items into the view in order, each at the next delivery +/// sequence, advancing the positions with each, and produce its stream +/// row. +pub(crate) fn stream_rows( + items: &[Item<'_>], + view: &mut RunView, + positions: &mut Positions, + stream_seq: &mut u64, +) -> Result, ProjectError> { + let mut rows = Vec::with_capacity(items.len()); + for item in items { + *stream_seq += 1; + view.fold(item, *stream_seq); + let row = match item { + Item::Petri(event) => { + positions.advance(event.id); + StreamRow { + stream_seq: *stream_seq, + item_kind: "petri", + item_id: event_id_text(&event.id), + event_json: serde_json::to_string(event).map_err(ProjectError::Encode)?, + } + } + Item::Platform(record) => { + positions.platform_seq = record.seq; + StreamRow { + stream_seq: *stream_seq, + item_kind: "platform", + item_id: record.seq.to_string(), + event_json: serde_json::to_string(record).map_err(ProjectError::Encode)?, + } + } + }; + rows.push(row); + } + Ok(rows) +} + +/// The run's stream past the cursor, read from the view tables: up to +/// `limit` rows with `stream_seq > after`, in order, in Fabro's envelope. +pub(super) async fn stream_after( + views: &DbPool, + run_id: RunId, + after: u64, + limit: usize, +) -> Result, ProjectError> { + let rows: Vec<(i64, String, String, String)> = sqlx::query_as( + "SELECT stream_seq, item_kind, item_id, event_json FROM petri_stream WHERE run_id = ? AND \ + stream_seq > ? ORDER BY stream_seq LIMIT ?", + ) + .bind(run_id.to_string()) + .bind(column(after)) + .bind(i64::try_from(limit).unwrap_or(i64::MAX)) + .fetch_all(views) + .await + .map_err(ProjectError::Database)?; + rows.into_iter() + .map(|(stream_seq, item_kind, item_id, event_json)| { + let item: serde_json::Value = + serde_json::from_str(&event_json).map_err(ProjectError::Encode)?; + let kind = match item_kind.as_str() { + "platform" => RunStreamItemKind::Platform, + _ => RunStreamItemKind::Petri, + }; + let recorded_at = item + .get("recorded_at") + .and_then(serde_json::Value::as_u64) + .unwrap_or(0); + Ok(RunStreamItem { + run_id, + stream_seq: u64::try_from(stream_seq).unwrap_or(0), + kind, + id: item_id, + recorded_at, + item, + }) + }) + .collect() +} + +/// A Petri event id as the stream names it: `//`. +#[must_use] +pub(crate) fn event_id_text(id: &EventId) -> String { + format!("{}/{}/{}", log_text(&id.source), id.seq, id.index) +} + +pub(super) fn log_text(source: &EventSource) -> String { + match source { + EventSource::Coordinator => "coordinator".to_string(), + EventSource::Execution { execution } => format!("execution {execution}"), + } +} + +pub(super) fn column(value: u64) -> i64 { + i64::try_from(value).unwrap_or(i64::MAX) +} diff --git a/lib/components/fabro-petri/src/recovery.rs b/lib/components/fabro-petri/src/recovery.rs index 2c068ed86..ee62ec941 100644 --- a/lib/components/fabro-petri/src/recovery.rs +++ b/lib/components/fabro-petri/src/recovery.rs @@ -31,45 +31,41 @@ //! the worker's run reaches, so its target is deferred, and the worker's //! hooks read the same plan and apply it through the scope's environment //! at `scope_acquired`, before the first attempt runs there -//! ([`bring_sandbox_to`]). +//! ([`bring_to`]). use std::collections::BTreeMap; use std::path::PathBuf; use std::sync::Arc; -use fabro_checkpoint::author::GitAuthor; use fabro_store::platform_records::CheckpointRecord; use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition}; -use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace}; -use fabro_types::{RunId, SandboxProviderKind}; +use fabro_types::RunId; +use fabro_types::settings::run::RunNamespace; use petri_execution::host::{self, HostError}; use petri_execution::inspect::{self, ExecutionInspection, InspectError}; use petri_execution::{Access, InvocationId, RunKey, RunStore}; -use petri_runtime::executor::ExecEnv; use petri_store::StoreError; use tracing::info; -use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces}; +use crate::checkpoint::{ + CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunGitSettings, RunWorkspaces, Site, +}; use crate::platform_records::{PlatformRecordError, PlatformRecords}; use crate::workspace::{WorkspaceLookup, WorkspaceLookupError}; -/// What recovery needs: the run, where its workspaces are, its records. +/// What recovery needs: the run, where its workspaces are, its records, +/// and its Git settings. pub struct RecoveryRequest { - pub run_id: RunId, + pub run_id: RunId, /// The run directory Petri ran under (the run's `petri` scratch). - pub run_dir: PathBuf, - pub store: Arc, - pub records: Arc, - pub author: GitAuthor, - pub checkpoint: RunCheckpointSettings, - /// Whether the run's workspaces are on this host. - pub host_workspaces: bool, + pub run_dir: PathBuf, + pub store: Arc, + pub records: Arc, + pub git: RunGitSettings, } impl RecoveryRequest { - /// The request a run's settings give: its Git author, its checkpoint - /// settings, and whether its sandbox provider keeps workspaces on this - /// host. + /// The request a run's settings give. #[must_use] pub fn for_run( run_id: RunId, @@ -83,14 +79,7 @@ impl RecoveryRequest { run_dir, store, records, - author: settings - .git - .author - .as_ref() - .map(GitAuthor::from) - .unwrap_or_default(), - checkpoint: settings.checkpoint.clone(), - host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL, + git: RunGitSettings::from(settings), } } } @@ -172,12 +161,6 @@ pub enum RecoveryError { }, } -/// One live execution's last durable finish and the snapshot it names. -struct Target { - execution: u64, - key: CheckpointKey, -} - /// Decide how the run continues: the snapshot every live workspace must sit /// on, from the records and the snapshot repository, with a lost record /// reconciled from the repository. Nothing is touched. @@ -220,80 +203,14 @@ pub async fn plan( if let Some(failed) = checkpoint_failure(&inspection.executions) { return Ok(Plan::Failed { reason: failed }); } - - let lookup = WorkspaceLookup::new(Arc::clone(&store), key); - let recorded = recorded_checkpoints(records, run_id).await?; - - // The snapshot each live execution's workspace must sit on. A live - // execution is one whose log records no exit: `inspect_run` reports it - // as incomplete. - let mut candidates: BTreeMap> = BTreeMap::new(); - for execution in inspection - .executions - .iter() - .filter(|execution| execution.status == "incomplete") - { - let Some(target) = last_finish(execution) else { - continue; - }; - let owned = lookup - .of_invocation(execution.invocation) - .await - .map_err(RecoveryError::Lookup)?; - if owned.is_empty() { - continue; - } - let mut found = false; - for workspace in owned { - let sha = match recorded.get(&target.key) { - Some((recorded_workspace, sha)) - if recorded_workspace.as_deref().is_none_or(|w| w == workspace) => - { - Some(sha.clone()) - } - _ => { - let sha = workspaces - .find(&workspace, target.key) - .await - .map_err(|source| RecoveryError::Workspace { - workspace: workspace.clone(), - source, - })?; - if let Some(sha) = &sha { - reconcile_record(records, run_id, target.key, &workspace, sha).await?; - } - sha - } - }; - if let Some(sha) = sha { - found = true; - candidates - .entry(workspace) - .or_default() - .push((Target { ..target }, sha)); - } - } - if !found { - return Ok(Plan::Failed { - reason: format!( - "no checkpoint snapshot exists for the last durable finish of execution {} \ - (firing {} attempt {}); the run cannot resume on stale files", - target.execution, target.key.firing, target.key.attempt - ), - }); - } - } - - let mut targets = BTreeMap::new(); - for (workspace, candidates) in candidates { - let sha = newest(workspaces, &workspace, &candidates).await?; - let key = candidates - .iter() - .find(|(_, candidate)| *candidate == sha) - .map_or(candidates[0].0.key, |(target, _)| target.key); - targets.insert(workspace, RestoreTarget { key, sha }); - } - Ok(Plan::Resume { targets }) + let planner = Planner { + records, + run_id, + workspaces, + lookup: WorkspaceLookup::new(store, key), + recorded: recorded_checkpoints(records, run_id).await?, + }; + planner.targets(&inspection.executions).await } /// Decide how the run continues, and bring its host workspaces to their @@ -302,8 +219,8 @@ pub async fn recover(request: RecoveryRequest) -> Result Result Result { + records: &'a dyn PlatformRecords, + run_id: &'a RunId, + workspaces: &'a RunWorkspaces, + lookup: WorkspaceLookup, + /// The run's checkpoint records by key: the workspace they name and + /// the commit. + recorded: BTreeMap, String)>, +} + +impl Planner<'_> { + /// The snapshot each live execution's workspace must sit on. A live + /// execution is one whose log records no exit: `inspect_run` reports + /// it as incomplete. A workspace several live executions share is + /// brought to the newest of their snapshots. + async fn targets(&self, executions: &[ExecutionInspection]) -> Result { + let mut candidates: BTreeMap> = BTreeMap::new(); + for execution in executions + .iter() + .filter(|execution| execution.status == "incomplete") + { + let Some(key) = last_finish(execution) else { + continue; + }; + let owned = self + .lookup + .of_invocation(execution.invocation) + .await + .map_err(RecoveryError::Lookup)?; + if owned.is_empty() { + continue; + } + let mut found = false; + for workspace in owned { + if let Some(sha) = self.snapshot_of(&workspace, key).await? { + found = true; + candidates + .entry(workspace) + .or_default() + .push(Candidate { key, sha }); + } + } + if !found { + return Ok(Plan::Failed { + reason: format!( + "no checkpoint snapshot exists for the last durable finish of {key}; the \ + run cannot resume on stale files" + ), + }); + } + } + + let mut targets = BTreeMap::new(); + for (workspace, candidates) in candidates { + let target = self.newest(&workspace, &candidates).await?; + targets.insert(workspace, target); + } + Ok(Plan::Resume { targets }) + } + + /// The snapshot of `key` in `workspace`: the commit its record names, + /// when the record names this workspace or none; else the commit found + /// by its key in the workspace's snapshot repository or history, which + /// is then recorded again for the record the crash lost. `None` when no + /// snapshot exists. + async fn snapshot_of( + &self, + workspace: &str, + key: CheckpointKey, + ) -> Result, RecoveryError> { + if let Some((recorded_workspace, sha)) = self.recorded.get(&key) { + if recorded_workspace + .as_deref() + .is_none_or(|recorded| recorded == workspace) + { + return Ok(Some(sha.clone())); + } + } + let found = self + .workspaces + .find(workspace, key) + .await + .map_err(|source| RecoveryError::Workspace { + workspace: workspace.to_string(), + source, + })?; + if let Some(sha) = &found { + self.reconcile_record(key, workspace, sha).await?; + } + Ok(found) + } + + /// Write the record a crash lost, from the commit found by its key. + async fn reconcile_record( + &self, + key: CheckpointKey, + workspace: &str, + sha: &str, + ) -> Result<(), RecoveryError> { + info!( + run_id = %self.run_id, + execution = key.execution, + firing = key.firing, + attempt = key.attempt, + sha, + "checkpoint record reconciled from the run branch" + ); + let record = PlatformRecord::Checkpoint(CheckpointRecord { + execution: key.execution, + firing: key.firing, + attempt: Some(key.attempt), + workspace: Some(workspace.to_string()), + git_commit_sha: Some(sha.to_string()), + diff_summary: None, + patch_blob: None, + operation: Some(key.operation()), + }); + self.records + .append( + self.run_id, + &record, + Some(StagePosition { + execution: key.execution, + firing: key.firing, + }), + ) + .await + .map_err(RecoveryError::Records)?; + Ok(()) + } + + /// Of the snapshots live executions name on one workspace, the one + /// every other descends from, else the last named. Two executions that + /// name the same commit share it under the first one's key. + async fn newest( + &self, + workspace: &str, + candidates: &[Candidate], + ) -> Result { + let mut chosen = &candidates[0]; + for candidate in &candidates[1..] { + if self + .workspaces + .is_ancestor(workspace, &chosen.sha, &candidate.sha) + .await + .map_err(|source| RecoveryError::Workspace { + workspace: workspace.to_string(), + source, + })? + { + chosen = candidate; + } + } + let chosen = candidates + .iter() + .find(|candidate| candidate.sha == chosen.sha) + .unwrap_or(chosen); + Ok(RestoreTarget { + key: chosen.key, + sha: chosen.sha.clone(), + }) + } +} + /// The reason a run with a failed checkpoint is reported failed, when it /// has one. fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option { @@ -368,16 +463,14 @@ fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option { }) } -/// The last `StepFinished` of an execution's log. -fn last_finish(execution: &ExecutionInspection) -> Option { +/// The last `StepFinished` of an execution's log, as the key of its +/// snapshot. +fn last_finish(execution: &ExecutionInspection) -> Option { let attempt = execution.engine.as_ref()?.attempts.last()?; - Some(Target { + Some(CheckpointKey { execution: execution.execution.raw(), - key: CheckpointKey { - execution: execution.execution.raw(), - firing: attempt.firing, - attempt: attempt.attempt, - }, + firing: attempt.firing, + attempt: attempt.attempt, }) } @@ -407,75 +500,14 @@ async fn recorded_checkpoints( Ok(recorded) } -/// Write the record a crash lost, from the commit found by its key. -async fn reconcile_record( - records: &dyn PlatformRecords, - run_id: &RunId, - key: CheckpointKey, - workspace: &str, - sha: &str, -) -> Result<(), RecoveryError> { - info!( - run_id = %run_id, - execution = key.execution, - firing = key.firing, - attempt = key.attempt, - sha, - "checkpoint record reconciled from the run branch" - ); - let record = PlatformRecord::Checkpoint(CheckpointRecord { - execution: key.execution, - firing: key.firing, - attempt: Some(key.attempt), - workspace: Some(workspace.to_string()), - git_commit_sha: Some(sha.to_string()), - diff_summary: None, - patch_blob: None, - operation: Some(key.operation()), - }); - records - .append( - run_id, - &record, - Some(StagePosition { - execution: key.execution, - firing: key.firing, - }), - ) - .await - .map_err(RecoveryError::Records)?; - Ok(()) -} - -/// Of the snapshots live executions name on one workspace, the one every -/// other descends from, else the last named. -async fn newest( - workspaces: &RunWorkspaces, - workspace: &str, - targets: &[(Target, String)], -) -> Result { - let mut chosen = &targets[0].1; - for (_, sha) in &targets[1..] { - if workspaces - .is_ancestor(workspace, chosen, sha) - .await - .map_err(|source| RecoveryError::Workspace { - workspace: workspace.to_string(), - source, - })? - { - chosen = sha; - } - } - Ok(chosen.clone()) -} - -/// Verify, reset or restore the host workspace onto its target: a +/// Verify, reset or restore the workspace at `site` onto its target: a /// workspace that still holds the commit is verified or reset in place; a -/// gone one, or a fresh directory with no history (a fork's first -/// acquisition), is restored from the snapshot repository. -pub async fn bring_host_to( +/// gone one, a fresh directory with no history (a fork's first +/// acquisition), or a sandbox whose repository lost the commit, is restored +/// from the snapshot repository (a bundle of the snapshot, into a sandbox). +pub async fn bring_to( workspaces: &RunWorkspaces, + site: &Site, workspace: &str, target: &RestoreTarget, ) -> Result { @@ -484,64 +516,22 @@ pub async fn bring_host_to( source, }; if workspaces - .has_commit(workspace, &target.sha) + .has_commit(site, &target.sha) .await .map_err(failed)? { if workspaces - .matches(workspace, &target.sha) + .matches(site, &target.sha) .await .map_err(failed)? { return Ok(WorkspaceAction::Verified); } - workspaces - .reset(workspace, &target.sha) - .await - .map_err(failed)?; + workspaces.reset(site, &target.sha).await.map_err(failed)?; return Ok(WorkspaceAction::Reset); } workspaces - .restore(workspace, target.key, &target.sha) - .await - .map_err(failed)?; - Ok(WorkspaceAction::Restored) -} - -/// Verify, reset or restore a sandbox workspace onto its target, through -/// the scope's environment: a retained sandbox that still holds the commit -/// is verified or reset in place; a fresh one, or one whose repository -/// lost the commit, is restored from a bundle of the snapshot. -pub async fn bring_sandbox_to( - workspaces: &RunWorkspaces, - env: &Arc, - workspace: &str, - target: &RestoreTarget, -) -> Result { - let failed = |source| RecoveryError::Workspace { - workspace: workspace.to_string(), - source, - }; - if workspaces - .has_commit_in(env, &target.sha) - .await - .map_err(failed)? - { - if workspaces - .matches_in(env, &target.sha) - .await - .map_err(failed)? - { - return Ok(WorkspaceAction::Verified); - } - workspaces - .reset_in(env, &target.sha) - .await - .map_err(failed)?; - return Ok(WorkspaceAction::Reset); - } - workspaces - .restore_in(env, workspace, target.key, &target.sha) + .restore(site, workspace, target.key, &target.sha) .await .map_err(failed)?; Ok(WorkspaceAction::Restored) diff --git a/lib/components/fabro-petri/src/run_store.rs b/lib/components/fabro-petri/src/run_store.rs index d1eae1bdb..683e74f2f 100644 --- a/lib/components/fabro-petri/src/run_store.rs +++ b/lib/components/fabro-petri/src/run_store.rs @@ -55,13 +55,14 @@ use std::collections::HashMap; use std::error::Error; -use std::sync::{Arc, Mutex, MutexGuard, PoisonError, Weak}; +use std::sync::{Arc, Mutex, Weak}; use std::time::{SystemTime, UNIX_EPOCH}; use std::{fmt, mem, ptr}; use fabro_db::DbPool; use fabro_store::BlobStore; use fabro_types::BlobHash; +use fabro_util::sync; use petri_store::{ Access, Digest, ExecutionId, LogId, OwnerId, Record, RunKey, RunLogs, RunStore, StoreError, }; @@ -150,7 +151,7 @@ impl SqliteRunStore { if result.rows_affected() == 0 { return Err(self.shared.not_found(key)); } - lock(&self.shared.live).remove(key); + sync::lock(&self.shared.live).remove(key); debug!(run_id = %key, "Petri run lease released from outside"); Ok(()) } @@ -174,7 +175,7 @@ impl SqliteRunStore { /// The writer handle for `owner`, once the lease is taken: the live one /// when this owner already holds a handle here, else a new one. fn writer(&self, key: &RunKey, owner: OwnerId) -> Arc { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); if let Some(handle) = live.get(key).and_then(Weak::upgrade) { if handle.owner.as_ref() == Some(&owner) { return handle; @@ -214,7 +215,7 @@ impl Shared { /// Await every release a dropped handle spawned, so what follows sees /// the lease as the drops left it. async fn drain_releases(&self) { - let pending = mem::take(&mut *lock(&self.releases)); + let pending = mem::take(&mut *sync::lock(&self.releases)); for release in pending { // A release task never panics: it reports its own failure. let _ = release.await; @@ -409,7 +410,7 @@ impl Drop for SqliteRunLogs { return; }; { - let mut live = lock(&self.shared.live); + let mut live = sync::lock(&self.shared.live); let this: *const Self = self; if live .get(&self.key) @@ -425,7 +426,7 @@ impl Drop for SqliteRunLogs { let release = runtime.spawn(async move { shared.release_owner(&key, &owner).await; }); - lock(&self.shared.releases).push(release); + sync::lock(&self.shared.releases).push(release); } Err(_) => { warn!( @@ -558,10 +559,6 @@ impl RunLogs for SqliteRunLogs { } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - /// Milliseconds since the Unix epoch, as SQLite stores them. fn now_ms() -> i64 { SystemTime::now() diff --git a/lib/components/fabro-petri/src/test_support.rs b/lib/components/fabro-petri/src/test_support.rs index 7174cfbc5..91793f60e 100644 --- a/lib/components/fabro-petri/src/test_support.rs +++ b/lib/components/fabro-petri/src/test_support.rs @@ -1,23 +1,35 @@ //! Petri's test kit, for Fabro crates that check a store implementation -//! against Petri's contract from their own tests, an in-memory platform +//! against Petri's contract from their own tests; an in-memory platform //! record store and an in-memory blob table for tests of the hooks and -//! recovery. Compiled only with the `test-support` feature, which a +//! recovery; and the readers over the view tables a test compares a live +//! view with. Compiled only with the `test-support` feature, which a //! dev-dependency turns on. -use std::collections::HashMap; -use std::sync::{Mutex, MutexGuard, PoisonError}; +use std::collections::{BTreeSet, HashMap}; +use std::sync::Mutex; use std::time::Duration; use async_trait::async_trait; use bytes::Bytes; -use fabro_store::platform_records::now_ms; -use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord}; +use fabro_db::DbPool; +use fabro_store::platform_records::{PlatformRecordStore, now_ms}; +use fabro_store::{ + PlatformRecord, PlatformRecordKind, RunProjection, StagePosition, StoredPlatformRecord, +}; use fabro_types::{BlobHash, RunId}; +use fabro_util::error::collect_chain; +use fabro_util::sync; +use petri_execution::events::{self, RunEvent}; +use petri_execution::{Access, CoordinatorEvent, RunKey, RunStore as _}; +use petri_store::StoreError; pub use petri_testkit::run_store; +use tracing::warn; +use crate::SqliteRunStore; use crate::blobs::Blobs; use crate::platform_records::{PlatformRecordError, PlatformRecords}; -use crate::projector::Projector; +use crate::projection::RunView; +use crate::projector::{self, Positions, ProjectError, Projector, order, stream}; /// Whether the projector keeps a cache for the run: the replay and the /// view its passes continue from. @@ -47,7 +59,7 @@ impl MemoryBlobs { /// How many blobs the table holds. #[must_use] pub fn len(&self) -> usize { - lock(&self.rows).len() + sync::lock(&self.rows).len() } #[must_use] @@ -60,12 +72,12 @@ impl MemoryBlobs { impl Blobs for MemoryBlobs { async fn write(&self, bytes: &[u8]) -> anyhow::Result { let hash = BlobHash::new(bytes); - lock(&self.rows).insert(hash, bytes.to_vec()); + sync::lock(&self.rows).insert(hash, bytes.to_vec()); Ok(hash) } async fn read(&self, hash: &BlobHash) -> anyhow::Result> { - Ok(lock(&self.rows) + Ok(sync::lock(&self.rows) .get(hash) .map(|bytes| Bytes::copy_from_slice(bytes))) } @@ -86,14 +98,13 @@ impl MemoryPlatformRecords { /// Every record of the run, in seq order. #[must_use] pub fn records(&self, run_id: &RunId) -> Vec { - lock(&self.runs).get(run_id).cloned().unwrap_or_default() + sync::lock(&self.runs) + .get(run_id) + .cloned() + .unwrap_or_default() } } -fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { - mutex.lock().unwrap_or_else(PoisonError::into_inner) -} - #[async_trait] impl PlatformRecords for MemoryPlatformRecords { async fn append( @@ -102,7 +113,7 @@ impl PlatformRecords for MemoryPlatformRecords { record: &PlatformRecord, position: Option, ) -> Result { - let mut runs = lock(&self.runs); + let mut runs = sync::lock(&self.runs); let records = runs.entry(*run_id).or_default(); let stored = StoredPlatformRecord { seq: records.len() as u64 + 1, @@ -126,3 +137,100 @@ impl PlatformRecords for MemoryPlatformRecords { .collect()) } } + +/// The run's projection rebuilt from its records alone, with nothing +/// stored: what a fresh projector would commit over the same records. A test +/// compares it with the live view. `records` and `views` are the two pools +/// [`Projector::new`] takes. +pub async fn rebuild( + records: &DbPool, + views: &DbPool, + run_id: RunId, +) -> Result<(Option, Positions, u64), ProjectError> { + let store = SqliteRunStore::new(records.clone()); + let platform = PlatformRecordStore::new(views.clone()); + let key = RunKey::new(run_id.to_string()); + let platform_records = platform.read(&run_id).await.map_err(ProjectError::Store)?; + let events = match store.open(&key, Access::Read).await { + Ok(logs) => events::replay_run(&*logs) + .await + .inspect_err(|error| { + warn!(error = %collect_chain(error).join(": "), "rebuild: the run does not replay"); + }) + .unwrap_or_default(), + Err(StoreError::NotFound { .. }) => Vec::new(), + Err(error) => return Err(ProjectError::Open(error)), + }; + let run_finished = events.iter().any(|event| { + matches!( + event.coordinator(), + Some(CoordinatorEvent::RunFinished { .. }) + ) + }); + let (items, _held) = + order::order_items(&events, &platform_records, &BTreeSet::new(), run_finished); + let mut view = RunView::new(); + let mut positions = Positions::default(); + let mut stream_seq = 0; + stream::stream_rows(&items, &mut view, &mut positions, &mut stream_seq)?; + Ok((view.projection, positions, stream_seq)) +} + +/// The stored view's positions and stream sequence; `views` is the pool +/// the view tables live in. +pub async fn stored_positions( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + projector::stored_positions(views, run_id).await +} + +/// The stored view's projection, for a test or a reader outside the store. +pub async fn stored_projection( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + let json: Option = + sqlx::query_scalar("SELECT projection_json FROM petri_projection WHERE run_id = ?") + .bind(run_id.to_string()) + .fetch_optional(views) + .await + .map_err(ProjectError::Database)?; + json.map(|json| serde_json::from_str(&json).map_err(ProjectError::Encode)) + .transpose() +} + +/// The stream rows of a run: `(stream_seq, item_kind, item_id)`, in order. +pub async fn stored_stream( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + let rows: Vec<(i64, String, String)> = sqlx::query_as( + "SELECT stream_seq, item_kind, item_id FROM petri_stream WHERE run_id = ? ORDER BY stream_seq", + ) + .bind(run_id.to_string()) + .fetch_all(views) + .await + .map_err(ProjectError::Database)?; + Ok(rows + .into_iter() + .map(|(seq, kind, id)| (u64::try_from(seq).unwrap_or(0), kind, id)) + .collect()) +} + +/// Every stored platform record of a run, for a reader outside the store. +pub async fn stored_platform_records( + views: &DbPool, + run_id: RunId, +) -> Result, ProjectError> { + PlatformRecordStore::new(views.clone()) + .read(&run_id) + .await + .map_err(ProjectError::Store) +} + +/// A recorded event's projection is what `RunEvent` serializes to. +#[must_use] +pub fn event_json(event: &RunEvent) -> serde_json::Value { + serde_json::to_value(event).unwrap_or_default() +} diff --git a/lib/components/fabro-petri/tests/hooks.rs b/lib/components/fabro-petri/tests/hooks.rs index 1ba189a95..5849b4633 100644 --- a/lib/components/fabro-petri/tests/hooks.rs +++ b/lib/components/fabro-petri/tests/hooks.rs @@ -24,7 +24,9 @@ use fabro_checkpoint::author::GitAuthor; use fabro_petri::admission::AdmittedGraphs; use fabro_petri::blobs::Blobs; use fabro_petri::check::{self, Bundle, CheckRequest, Launch}; -use fabro_petri::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces}; +use fabro_petri::checkpoint::{ + CHECKPOINT_FAILED_CLASS, CheckpointKey, RunGitSettings, RunWorkspaces, +}; use fabro_petri::controls::RunControls; use fabro_petri::engine::{self, Execution, RunRequest, RunStatus}; use fabro_petri::hooks::HooksSpec; @@ -166,13 +168,13 @@ impl Harness { fn hooks(&self, provider: &SandboxProviderKind) -> HooksSpec { HooksSpec { - records: Arc::clone(&self.records) as Arc, - author: GitAuthor::default(), - identity_source: GitIdentitySource::Default, - checkpoint: RunCheckpointSettings::default(), - artifacts: self.artifacts.clone(), - host_workspaces: *provider == SandboxProviderKind::LOCAL, - test_gates: None, + records: Arc::clone(&self.records) as Arc, + git: RunGitSettings { + host_workspaces: *provider == SandboxProviderKind::LOCAL, + ..RunGitSettings::default() + }, + artifacts: self.artifacts.clone(), + test_gates: None, } } @@ -285,13 +287,11 @@ impl Harness { async fn recover(&self) -> Recovery { recovery::recover(RecoveryRequest { - run_id: self.run_id, - run_dir: self.run_dir.clone(), - store: Arc::clone(&self.store) as Arc, - records: Arc::clone(&self.records) as Arc, - author: GitAuthor::default(), - checkpoint: RunCheckpointSettings::default(), - host_workspaces: true, + run_id: self.run_id, + run_dir: self.run_dir.clone(), + store: Arc::clone(&self.store) as Arc, + records: Arc::clone(&self.records) as Arc, + git: RunGitSettings::default(), }) .await .expect("recovery decides") diff --git a/lib/components/fabro-petri/tests/projection.rs b/lib/components/fabro-petri/tests/projection.rs index 31f6e6ca1..aa7ff9914 100644 --- a/lib/components/fabro-petri/tests/projection.rs +++ b/lib/components/fabro-petri/tests/projection.rs @@ -418,15 +418,15 @@ fn json(value: &T) -> serde_json::Value { /// The stored view equals the view rebuilt from the records alone: the /// projection, the positions and the delivery sequence. async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { - let stored = projector::stored_projection(pool, run_id) + let stored = petri_support::stored_projection(pool, run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection"); - let (stored_positions, stored_stream_seq) = projector::stored_positions(pool, run_id) + let (stored_positions, stored_stream_seq) = petri_support::stored_positions(pool, run_id) .await .expect("the positions read") .expect("the run has positions"); - let (rebuilt, positions, stream_seq) = projector::rebuild(pool, pool, run_id) + let (rebuilt, positions, stream_seq) = petri_support::rebuild(pool, pool, run_id) .await .expect("the run rebuilds"); let rebuilt = rebuilt.expect("the rebuild has a projection"); @@ -442,7 +442,7 @@ async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { positions.petri.sort(); assert_eq!(stored_positions, positions); assert_eq!(stored_stream_seq, stream_seq); - let stream = projector::stored_stream(pool, run_id) + let stream = petri_support::stored_stream(pool, run_id) .await .expect("the stream reads"); let seqs: Vec = stream.iter().map(|(seq, _, _)| *seq).collect(); @@ -454,7 +454,7 @@ async fn assert_view_equals_rebuild(pool: &DbPool, run_id: RunId) { } async fn stage_states(pool: &DbPool, run_id: RunId) -> Vec<(String, StageState)> { - let stored = projector::stored_projection(pool, run_id) + let stored = petri_support::stored_projection(pool, run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection"); @@ -472,7 +472,7 @@ async fn the_hello_bundle_projects_live_as_it_rebuilds() { let scenario = hello_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -501,7 +501,7 @@ async fn a_large_output_projects_as_its_blob_reference() { let scenario = large_output_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -525,7 +525,7 @@ async fn a_command_workflow_projects_live_as_it_rebuilds() { let scenario = command_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -554,7 +554,7 @@ async fn a_parallel_workflow_projects_its_branches_as_child_executions() { let scenario = parallel_scenario().await; run_live(&scenario).await; assert_view_equals_rebuild(&scenario.pool, scenario.run_id).await; - let stored = projector::stored_projection(&scenario.pool, scenario.run_id) + let stored = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -593,7 +593,7 @@ async fn dropped_wake_ups_are_caught_up_by_the_next_signal() { let scenario = command_scenario().await; run_unobserved(&scenario).await; assert!( - projector::stored_projection(&scenario.pool, scenario.run_id) + petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .is_none(), @@ -780,12 +780,13 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( .expect("the first pass commits"); assert!(!pass.skipped); assert!(!pass.health.complete, "the run has not finished"); - let (positions_before, stream_before) = projector::stored_positions(&replayed, scenario.run_id) - .await - .expect("reads") - .expect("positions"); + let (positions_before, stream_before) = + petri_support::stored_positions(&replayed, scenario.run_id) + .await + .expect("reads") + .expect("positions"); assert_eq!(pass.stream_seq, stream_before); - let stream_rows_before = projector::stored_stream(&replayed, scenario.run_id) + let stream_rows_before = petri_support::stored_stream(&replayed, scenario.run_id) .await .expect("reads") .len(); @@ -801,7 +802,7 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( "{crashed:?}" ); assert_eq!( - projector::stored_positions(&replayed, scenario.run_id) + petri_support::stored_positions(&replayed, scenario.run_id) .await .expect("reads") .expect("positions"), @@ -813,11 +814,12 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( let after = Projector::new(replayed.clone(), replayed.clone()); let report = after.startup_pass().await.expect("the restart catches up"); assert_eq!((report.runs, report.projected), (1, 1)); - let (positions_after, stream_after) = projector::stored_positions(&replayed, scenario.run_id) - .await - .expect("reads") - .expect("positions"); - let stream_rows_after = projector::stored_stream(&replayed, scenario.run_id) + let (positions_after, stream_after) = + petri_support::stored_positions(&replayed, scenario.run_id) + .await + .expect("reads") + .expect("positions"); + let stream_rows_after = petri_support::stored_stream(&replayed, scenario.run_id) .await .expect("reads"); // Only the suffix was applied: the stream grew by the suffix's events, @@ -845,11 +847,11 @@ async fn a_crash_between_the_record_commit_and_the_view_applies_only_the_suffix( // And the copy agrees with the run projected in one go over the source. let source = Projector::new(scenario.pool.clone(), scenario.pool.clone()); source.startup_pass().await.expect("the source projects"); - let whole = projector::stored_projection(&scenario.pool, scenario.run_id) + let whole = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); - let pieced = projector::stored_projection(&replayed, scenario.run_id) + let pieced = petri_support::stored_projection(&replayed, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -904,11 +906,11 @@ async fn a_restarted_projector_agrees_over_nested_child_executions() { let whole = Projector::new(scenario.pool.clone(), scenario.pool.clone()); whole.startup_pass().await.expect("the source projects"); - let one_go = projector::stored_projection(&scenario.pool, scenario.run_id) + let one_go = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); - let restarted = projector::stored_projection(&staged, scenario.run_id) + let restarted = petri_support::stored_projection(&staged, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -937,11 +939,11 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { .await .expect("the clean pass commits"); assert!(clean.health.complete, "{:?}", clean.health.incomplete); - let (positions, stream_seq) = projector::stored_positions(&scenario.pool, scenario.run_id) + let (positions, stream_seq) = petri_support::stored_positions(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("positions"); - let before = projector::stored_projection(&scenario.pool, scenario.run_id) + let before = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -973,7 +975,7 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { held.health ); let (positions_after, stream_after) = - projector::stored_positions(&scenario.pool, scenario.run_id) + petri_support::stored_positions(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("positions"); @@ -982,7 +984,7 @@ async fn a_torn_tail_holds_the_view_and_reports_the_run_incomplete() { "the view did not advance past the tear" ); assert_eq!(stream_after, stream_seq); - let after = projector::stored_projection(&scenario.pool, scenario.run_id) + let after = petri_support::stored_projection(&scenario.pool, scenario.run_id) .await .expect("reads") .expect("stored"); @@ -1240,9 +1242,10 @@ impl GateRun { async fn pending(&self) -> fabro_types::RunProjection { let deadline = Instant::now() + Duration::from_secs(30); loop { - let stored = projector::stored_projection(&self.scenario.pool, self.scenario.run_id) - .await - .expect("the stored projection reads"); + let stored = + petri_support::stored_projection(&self.scenario.pool, self.scenario.run_id) + .await + .expect("the stored projection reads"); if let Some(stored) = stored.filter(|stored| !stored.pending_interviews.is_empty()) { return stored; } @@ -1255,7 +1258,7 @@ impl GateRun { } async fn stored(&self) -> fabro_types::RunProjection { - projector::stored_projection(&self.scenario.pool, self.scenario.run_id) + petri_support::stored_projection(&self.scenario.pool, self.scenario.run_id) .await .expect("the stored projection reads") .expect("the run has a stored projection") @@ -1282,7 +1285,7 @@ async fn an_expired_question_is_pending_while_the_gate_waits_and_closes_on_the_e let run_id = gate.scenario.run_id; let deadline = Instant::now() + Duration::from_secs(30); loop { - let stored = projector::stored_projection(&pool, run_id) + let stored = petri_support::stored_projection(&pool, run_id) .await .expect("the stored projection reads"); if let Some(stored) = stored.filter(|stored| !stored.pending_interviews.is_empty()) { diff --git a/lib/foundation/fabro-types/src/conclusion.rs b/lib/foundation/fabro-types/src/conclusion.rs index a06697070..65ec19c23 100644 --- a/lib/foundation/fabro-types/src/conclusion.rs +++ b/lib/foundation/fabro-types/src/conclusion.rs @@ -40,3 +40,27 @@ pub struct Conclusion { #[serde(default)] pub diff: RunDiff, } + +impl Conclusion { + /// A conclusion that records only how the run ended: no timing, stages, + /// usage or diff. What a terminal lifecycle record gives when the + /// engine recorded no finish of its own. + #[must_use] + pub fn outcome_only( + timestamp: DateTime, + status: StageOutcome, + failure: Option, + ) -> Self { + Self { + timestamp, + status, + timing: RunTiming::default(), + failure, + final_git_commit_sha: None, + stages: Vec::new(), + usage: None, + total_retries: 0, + diff: RunDiff::default(), + } + } +} diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs index 339aed897..89a884ff5 100644 --- a/lib/foundation/fabro-types/src/run_projection.rs +++ b/lib/foundation/fabro-types/src/run_projection.rs @@ -122,6 +122,19 @@ impl StageModelUsage { pub const MODE_AGENT: &'static str = "agent"; pub const MODE_ACP: &'static str = "acp"; + /// The usage record of a stage that named its provider and model, with + /// no request controls. + #[must_use] + pub fn new(mode: &str, provider: Option, model: Option) -> Self { + Self { + mode: mode.to_string(), + provider, + model, + reasoning_effort: None, + speed: None, + } + } + /// Build the usage record from a `stage.prompt` event, returning `None` /// when the event carried no model metadata. #[must_use] diff --git a/lib/foundation/fabro-types/src/status.rs b/lib/foundation/fabro-types/src/status.rs index 2b2c63e75..1a7db7061 100644 --- a/lib/foundation/fabro-types/src/status.rs +++ b/lib/foundation/fabro-types/src/status.rs @@ -120,6 +120,27 @@ impl RunStatus { } } + /// The status a pause takes the run to: paused, remembering the block + /// the run was under so the unpause can restore it. + #[must_use] + pub fn paused(self) -> Self { + Self::Paused { + prior_block: self.blocked_reason(), + } + } + + /// The status an unpause takes the run to: back to the block the pause + /// remembered, else running. + #[must_use] + pub fn unpaused(self) -> Self { + match self { + Self::Paused { + prior_block: Some(blocked_reason), + } => Self::Blocked { blocked_reason }, + _ => Self::Running, + } + } + pub fn terminal_status(self) -> Option { match self { Self::Succeeded { reason } => Some(TerminalStatus::Succeeded { reason }), diff --git a/lib/foundation/fabro-util/src/lib.rs b/lib/foundation/fabro-util/src/lib.rs index c759f7171..28d4545e3 100644 --- a/lib/foundation/fabro-util/src/lib.rs +++ b/lib/foundation/fabro-util/src/lib.rs @@ -12,6 +12,7 @@ pub mod printer; pub mod run_log; pub mod session_secret; pub mod shell; +pub mod sync; pub mod terminal; pub mod text; pub mod time; diff --git a/lib/foundation/fabro-util/src/sync.rs b/lib/foundation/fabro-util/src/sync.rs new file mode 100644 index 000000000..851acdd4a --- /dev/null +++ b/lib/foundation/fabro-util/src/sync.rs @@ -0,0 +1,11 @@ +//! Locking helpers shared across crates. + +use std::sync::{Mutex, MutexGuard, PoisonError}; + +/// Lock a mutex, recovering the guard when another holder panicked. The +/// state such a mutex guards is bookkeeping (a cache, a set of ids, a +/// counter) that stays usable after a panic elsewhere, so the poison is +/// cleared rather than propagated. +pub fn lock(mutex: &Mutex) -> MutexGuard<'_, T> { + mutex.lock().unwrap_or_else(PoisonError::into_inner) +}