Merge branch 'petri-integration-hooks' into petri-integration

# Conflicts:
#	Cargo.lock
#	lib/apps/fabro-cli/src/commands/run/petri_worker.rs
#	lib/apps/fabro-cli/tests/it/scenario/petri.rs
#	lib/apps/fabro-server/src/server/petri_runs.rs
#	lib/components/fabro-petri/Cargo.toml
#	lib/components/fabro-petri/src/engine.rs
#	lib/components/fabro-petri/src/lib.rs
#	lib/components/fabro-store/src/platform_records.rs
This commit is contained in:
Bryan Helmkamp 2026-09-18 00:51:21 -04:00
commit c9c4c27753
No known key found for this signature in database
29 changed files with 4329 additions and 43 deletions

2
Cargo.lock generated
View file

@ -2894,11 +2894,13 @@ dependencies = [
"chrono",
"fabro-api",
"fabro-auth",
"fabro-checkpoint",
"fabro-client",
"fabro-db",
"fabro-http",
"fabro-interview",
"fabro-llm",
"fabro-petri",
"fabro-store",
"fabro-test",
"fabro-types",

View file

@ -3418,6 +3418,59 @@ paths:
schema:
$ref: "#/components/schemas/ErrorResponse"
/api/v1/runs/{id}/petri/platform-records:
get:
operationId: listPetriPlatformRecords
tags: [Run Internals]
summary: List Petri Platform Records
description: |
The run's platform records (Fabro's own facts about a Petri run: a
checkpoint commit, a pull request, a notification), in `seq` order,
optionally of one kind. What a run's worker reads to find an effect
it already performed before performing it again.
parameters:
- $ref: "#/components/parameters/RunId"
- $ref: "#/components/parameters/PetriPlatformRecordKind"
responses:
"200":
description: The run's platform records
content:
application/json:
schema:
$ref: "#/components/schemas/PetriPlatformRecordList"
post:
operationId: appendPetriPlatformRecord
tags: [Run Internals]
summary: Append Petri Platform Record
description: |
Stores one platform record at the run's next `seq`, tied to the
Petri stage named by `execution` and `firing` when it belongs to one.
The record is the JSON of a Fabro platform record, tagged by `kind`.
parameters:
- $ref: "#/components/parameters/RunId"
requestBody:
required: true
content:
application/json:
schema:
$ref: "#/components/schemas/PetriPlatformRecordAppendRequest"
responses:
"200":
description: The record as stored
content:
application/json:
schema:
$ref: "#/components/schemas/PetriPlatformRecord"
"400":
description: The record is not a platform record
headers:
x-request-id:
$ref: "#/components/headers/XRequestId"
content:
application/json:
schema:
$ref: "#/components/schemas/ErrorResponse"
/api/v1/runs/{id}/stages/{stageId}/logs/output:
get:
operationId: getRunStageCommandLog
@ -6222,6 +6275,17 @@ components:
type: string
example: 18f3c2a9e1b4-42017-0-9f3a1c7e2b5d
PetriPlatformRecordKind:
name: kind
in: query
required: false
description: >-
Only the platform records of this kind, as its `kind` tag spells it
(`checkpoint`, `pull_request.created`, ...).
schema:
type: string
example: checkpoint
ArtifactFilename:
name: filename
in: query
@ -11000,6 +11064,71 @@ components:
items:
$ref: "#/components/schemas/PetriRecord"
PetriPlatformRecord:
description: >-
One of Fabro's platform records of a Petri run, as stored: the
record's JSON tagged by `kind`, its position in the run's platform
record sequence, and the Petri stage it belongs to when it belongs
to one.
type: object
required:
- seq
- recorded_at
- record
properties:
seq:
type: integer
format: uint64
description: The record's position in the run's platform records, from 1.
example: 4
recorded_at:
type: integer
format: uint64
description: Milliseconds since the Unix epoch when the record was stored.
example: 1758067200123
record:
type: object
additionalProperties: true
description: The platform record itself, tagged by `kind`.
execution:
type: integer
format: uint64
description: The Petri execution the record belongs to, with `firing`.
firing:
type: integer
format: uint64
description: The Petri firing the record belongs to, with `execution`.
PetriPlatformRecordAppendRequest:
description: One platform record to store for the run.
type: object
required:
- record
properties:
record:
type: object
additionalProperties: true
description: The platform record, tagged by `kind`.
execution:
type: integer
format: uint64
description: The Petri execution the record belongs to, with `firing`.
firing:
type: integer
format: uint64
description: The Petri firing the record belongs to, with `execution`.
PetriPlatformRecordList:
description: The run's platform records, in `seq` order.
type: object
required:
- records
properties:
records:
type: array
items:
$ref: "#/components/schemas/PetriPlatformRecord"
CommandTermination:
description: Terminal state for a command execution.
type: string

View file

@ -27,6 +27,10 @@
//! for good cancels the run the same way, and the worker exits with that
//! loss as its error once the run has settled.
//!
//! Fabro's hooks ride the run with their platform records over the same
//! client: the checkpoint commit in the run's host workspace before every
//! durable finish, and its record after every route.
//!
//! The runtime's settings layer is left empty here: the run's graphs were
//! lowered and admitted at create time with the server's layer, and nothing
//! lowers again at execution. The model client is built from the worker's
@ -47,11 +51,14 @@ use fabro_interview::ControlInterviewer;
use fabro_llm::credentials::{CredentialProvider, readiness};
use fabro_petri::blobs::ClientBlobs;
use fabro_petri::engine::{self, Conclusion, Execution, RunRequest};
use fabro_petri::hooks::HooksSpec;
use fabro_petri::interview::{Approval, EventSinkQuestions, FabroInterviewer};
use fabro_petri::petri::OwnerId;
use fabro_petri::platform_records::HttpPlatformRecords;
use fabro_petri::runtime::{self, RuntimeSpec};
use fabro_petri::secrets::VaultSecrets;
use fabro_petri::{HttpRunStore, admission};
use fabro_static::EnvVars;
use fabro_store::RunProjection;
use fabro_types::settings::run::{ApprovalMode, RunMode};
use fabro_types::{FailureReason, RunId, RunTiming, StageOutcome, SuccessReason};
@ -166,6 +173,11 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
}
runner::set_worker_title(&run_id, WorkerTitlePhase::Running);
let hooks = HooksSpec::for_run(
Arc::new(HttpPlatformRecords::new(worker.client.clone_for_reuse())),
&worker.run_state.spec.settings.run,
)
.with_test_gates(test_checkpoint_gates());
let request = RunRequest {
run_id: run_id.to_string(),
run_dir: worker.run_dir.join("petri"),
@ -188,6 +200,7 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
worker.client.clone_for_reuse(),
run_id,
))),
hooks: Some(hooks),
};
let run = Box::pin(engine::run(request));
tokio::pin!(run);
@ -259,6 +272,15 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
}
}
/// A test's checkpoint gate directory, when the server forwarded one.
#[expect(
clippy::disallowed_methods,
reason = "the gate directory is a test-only process-env facade the server forwards by name"
)]
fn test_checkpoint_gates() -> Option<PathBuf> {
std::env::var_os(EnvVars::FABRO_TEST_CHECKPOINT_GATES).map(PathBuf::from)
}
/// The runtime the worker hands Petri: no settings layer (nothing lowers
/// at execution), the model client over the worker's catalog and vault for
/// the providers whose credentials resolve, the run's mode, and the Fabro

View file

@ -29,12 +29,13 @@ use std::time::{Duration, Instant};
use fabro_client::ServerTarget;
use fabro_config::{Storage, envfile};
use fabro_petri::SqliteRunStore;
use fabro_petri::checkpoint::CheckpointKey;
use fabro_petri::engine::{self, RunStatus};
use fabro_petri::petri::RunKey;
use fabro_static::EnvVars;
use fabro_store::EventEnvelope;
use fabro_store::{EventEnvelope, PlatformRecord, PlatformRecordKind, PlatformRecordStore};
use fabro_test::{apply_test_isolation, expect_reqwest_json, isolated_storage_dir, test_context};
use fabro_types::EventBody;
use fabro_types::{EventBody, RunId};
use crate::cmd::support::created_run_id;
use crate::support::{
@ -81,6 +82,8 @@ struct RunningServer {
config_path: PathBuf,
port: u16,
api_base_url: String,
/// The checkpoint gate directory the server forwards to its workers.
gates_dir: PathBuf,
}
impl RunningServer {
@ -103,6 +106,8 @@ impl RunningServer {
.expect("the server env writes");
fabro_util::dev_token::write_dev_token(&runtime_directory.dev_token_path(), TEST_DEV_TOKEN)
.expect("the dev token writes");
let gates_dir = home_root.path().join("checkpoint-gates");
std::fs::create_dir_all(&gates_dir).expect("the gates dir creates");
let mut server = Self {
child: None,
home_root,
@ -111,6 +116,7 @@ impl RunningServer {
config_path,
port,
api_base_url: format!("http://127.0.0.1:{port}"),
gates_dir,
};
server.launch().await;
server
@ -129,6 +135,7 @@ impl RunningServer {
EnvVars::FABRO_HOME,
self.home_root.path().join("fabro-home"),
);
cmd.env(EnvVars::FABRO_TEST_CHECKPOINT_GATES, &self.gates_dir);
cmd.args(["server", "start", "--foreground"])
.arg("--storage-dir")
.arg(&self.storage_dir)
@ -137,16 +144,16 @@ impl RunningServer {
.arg("--config")
.arg(&self.config_path)
.stdin(Stdio::null())
.stdout(Stdio::null())
.stdout(self.stderr_log())
.stderr(self.stderr_log());
let mut child = cmd.spawn().expect("the server spawns");
wait_for_http_ready(&self.api_base_url, &mut child).await;
self.child = Some(child);
}
/// Where the server's stderr goes: a file beside its storage, so a
/// chatty server never blocks on a pipe nobody reads, and a failing
/// test can show it.
/// Where the server's stdout and stderr go: a file beside its storage,
/// so a chatty server never blocks on a pipe nobody reads, and a
/// failing test can show its log.
fn stderr_log(&self) -> Stdio {
let path = self.storage_dir.with_file_name("server.stderr.log");
let file = std::fs::OpenOptions::new()
@ -209,6 +216,117 @@ impl RunningServer {
.expect("the server database opens");
SqliteRunStore::new(database.clone_pool())
}
/// The run's platform records in the server's database.
async fn platform_records(&self) -> PlatformRecordStore {
let database = fabro_db::Database::connect(Storage::new(&self.storage_dir).sqlite_path())
.await
.expect("the server database opens");
PlatformRecordStore::new(database.clone_pool())
}
/// Where the run's worker ran Petri: the run's scratch under the
/// server's storage.
fn petri_run_dir(&self, run_id: &str) -> PathBuf {
let run_id: RunId = run_id.parse().expect("the run id parses");
Storage::new(&self.storage_dir)
.run_scratch(&run_id)
.root()
.join("petri")
}
/// The worker's own log for the run.
fn worker_log(&self, run_id: &str) -> PathBuf {
let run_id: RunId = run_id.parse().expect("the run id parses");
Storage::new(&self.storage_dir)
.run_scratch(&run_id)
.root()
.join("runtime")
.join("server.log")
}
/// Hold the worker's checkpoint at `point` (`commit` or `record`) for
/// `node` until [`release`](Self::release).
fn hold(&self, point: &str, node: &str) {
std::fs::write(self.gates_dir.join(format!("{point}.{node}.hold")), "")
.expect("the hold file writes");
}
fn release(&self, point: &str, node: &str) {
std::fs::write(self.gates_dir.join(format!("{point}.{node}.release")), "")
.expect("the release file writes");
}
/// Wait until the worker's log says its checkpoint is held at a gate.
fn wait_until_held(&self, run_id: &str, point: &str, node: &str) {
let log = self.worker_log(run_id);
let needle = format!("checkpoint held at a test gate point=\"{point}\" node=\"{node}\"");
let deadline = Instant::now() + RUN_TIMEOUT;
loop {
let text = std::fs::read_to_string(&log).unwrap_or_default();
if text.contains(&needle) {
return;
}
assert!(
Instant::now() < deadline,
"the worker never held at {point}.{node}; log:\n{text}"
);
std::thread::sleep(POLL);
}
}
/// The one host workspace of the run, and the commits on its run
/// branch, oldest first, as `(subject, key)`.
fn workspace_commits(&self, run_id: &str) -> (PathBuf, Vec<(String, Option<CheckpointKey>)>) {
let scopes = self.petri_run_dir(run_id).join("scopes");
let mut workspaces: Vec<PathBuf> = std::fs::read_dir(&scopes)
.expect("the scopes directory lists")
.map(|entry| entry.expect("an entry reads").path().join("work"))
.collect();
assert_eq!(workspaces.len(), 1, "one workspace: {workspaces:?}");
let workspace = workspaces.remove(0);
let output = Command::new("git")
.args(["log", "--reverse", "--format=%s%x00%B%x1e"])
.current_dir(&workspace)
.output()
.expect("git runs");
assert!(
output.status.success(),
"git log failed: {}",
String::from_utf8_lossy(&output.stderr)
);
let log = String::from_utf8_lossy(&output.stdout).into_owned();
let commits = log
.split('\u{1e}')
.filter(|entry| !entry.trim().is_empty())
.map(|entry| {
let mut parts = entry.trim_start().splitn(2, '\0');
let subject = parts.next().unwrap_or_default().to_string();
let body = parts.next().unwrap_or_default();
(subject, CheckpointKey::from_message(body))
})
.collect();
(workspace, commits)
}
/// The run's checkpoint records, in seq order, as `(node position, sha)`.
async fn checkpoints(&self, run_id: &str) -> Vec<(CheckpointKey, String)> {
let run_id: RunId = run_id.parse().expect("the run id parses");
self.platform_records()
.await
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
.await
.expect("the checkpoint records read")
.into_iter()
.filter_map(|stored| match stored.record {
PlatformRecord::Checkpoint(record) => Some((
CheckpointKey::from_operation(record.operation.as_ref()?)?,
record.git_commit_sha?,
)),
_ => None,
})
.collect()
}
}
impl Drop for RunningServer {
@ -357,6 +475,28 @@ async fn run_events(server: &RunningServer, run_id: &str) -> Vec<EventEnvelope>
parse_event_envelopes(&run_json(server, &format!("runs/{run_id}/events")).await)
}
/// The run's events once `terminal` is among them: the events list is a
/// projection that settles after the run's status does.
async fn settled_events(
server: &RunningServer,
run_id: &str,
terminal: &str,
) -> Vec<EventEnvelope> {
let deadline = Instant::now() + RUN_TIMEOUT;
loop {
let events = run_events(server, run_id).await;
if event_names(&events).contains(&terminal) {
return events;
}
assert!(
Instant::now() < deadline,
"`{terminal}` was never projected for run {run_id}: {:?}",
event_names(&events)
);
tokio::time::sleep(POLL).await;
}
}
fn event_names(events: &[EventEnvelope]) -> Vec<&str> {
events
.iter()
@ -432,7 +572,7 @@ async fn a_petri_run_executes_in_the_server_launched_worker() {
let state = run_json(&server, &format!("runs/{run_id}/state")).await;
assert_eq!(state["spec"]["engine"]["kind"], "petri", "state: {state}");
let events = run_events(&server, &run_id).await;
let events = settled_events(&server, &run_id, "run.completed").await;
let names = event_names(&events);
assert_eq!(
names
@ -519,14 +659,14 @@ async fn a_petri_run_resumes_in_a_new_worker_after_the_server_restarts() {
std::fs::write(&gate, "go").expect("the gate opens");
let status = wait_for_status(&server, &run_id, &["succeeded", "failed"]).await;
let events = run_events(&server, &run_id).await;
let names = event_names(&events);
assert_eq!(
status,
"succeeded",
"events: {names:?}\nserver stderr:\n{}",
"server stderr:\n{}",
server.stderr_text()
);
let events = settled_events(&server, &run_id, "run.completed").await;
let names = event_names(&events);
assert_eq!(
names
.iter()
@ -811,3 +951,375 @@ async fn an_unanswered_gate_in_the_worker_expires_with_its_default() {
assert!(questions(&server, &run_id).await.is_empty());
server.shutdown();
}
/// A shell loop that waits for `gate` to exist.
fn wait_for(gate: &Path) -> String {
format!("while [ ! -f {} ]; do sleep 0.05; done", gate.display())
}
/// Three command stages: `one` writes a file, `two` writes another after
/// the gate opens (and logs each run), `three` checks both files.
fn three_stage_bundle(context: &fabro_test::TestContext, gate: &Path) -> PathBuf {
write_petri_workflow(
context,
&format!(
"digraph Stages {{\n graph [goal=\"Three stages\", default_max_retries=0]\n start \
[shape=Mdiamond]\n exit [shape=Msquare]\n one [shape=parallelogram, script=\"echo \
one > one.txt\"]\n two [shape=parallelogram, script=\"echo run >> two.log; {}; \
echo two > two.txt\"]\n three [shape=parallelogram, script=\"test \\\"$(cat \
one.txt)\\\" = one && test \\\"$(cat two.txt)\\\" = two && cp two.log \
three.log\"]\n start -> one -> two -> three -> exit\n}}\n",
wait_for(gate)
),
)
}
/// Kill the server first, so it never observes the worker exit, then the
/// worker's whole process group, then the stage's own process group when a
/// stage was waiting on `gate`: a stage process runs in a group of its own
/// under the host plugin, and a machine crash takes it with everything
/// else, where a killed worker alone would leave it writing into the
/// workspace.
fn crash(server: &mut RunningServer, worker: u32, gate: Option<&Path>) {
server.kill();
fabro_proc::sigkill_process_group(worker);
let deadline = Instant::now() + Duration::from_secs(10);
while fabro_proc::process_running(worker) {
assert!(Instant::now() < deadline, "the worker did not die");
std::thread::sleep(POLL);
}
let Some(gate) = gate else {
return;
};
let output = Command::new("pgrep")
.args(["-f", &gate.display().to_string()])
.output()
.expect("pgrep runs");
for pid in String::from_utf8_lossy(&output.stdout)
.lines()
.filter_map(|line| line.trim().parse::<u32>().ok())
{
fabro_proc::sigkill_process_group(pid);
}
let deadline = Instant::now() + Duration::from_secs(10);
while gate_is_polled(gate) {
assert!(Instant::now() < deadline, "the stage did not die");
std::thread::sleep(POLL);
}
}
/// Wait for the run to succeed after a restart, with the server's stderr
/// on failure.
async fn wait_for_success(server: &RunningServer, run_id: &str) {
let status = wait_for_status(server, run_id, &["succeeded", "failed"]).await;
assert_eq!(
status,
"succeeded",
"run: {}\nserver stderr:\n{}",
run_json(server, &format!("runs/{run_id}")).await,
server.stderr_text()
);
}
/// The subjects of the commits on the run branch.
fn subjects(commits: &[(String, Option<CheckpointKey>)]) -> Vec<&str> {
commits
.iter()
.map(|(subject, _)| subject.as_str())
.collect()
}
/// The commit subjects one run of the three-stage bundle produces.
fn three_stage_subjects(run_id: &str) -> Vec<String> {
["start", "one", "two", "three", "exit"]
.iter()
.map(|node| format!("fabro({run_id}): {node} (success)"))
.collect()
}
/// A worker killed after a stage's finish is durable: on the restart the
/// stage's commit is not repeated, the stage in flight reruns on the
/// snapshot (its partial output gone), and the next stage sees both.
#[tokio::test(flavor = "multi_thread")]
async fn a_crash_after_a_durable_finish_keeps_its_one_commit() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let gate = context.temp_dir.join("two.gate");
let workspace = three_stage_bundle(&context, &gate);
let run_id = run_detached(&context, &server, &workspace);
wait_for_status(&server, &run_id, &["running"]).await;
let worker = wait_for_worker(&run_id);
wait_until_gate_is_polled(&gate);
crash(&mut server, worker, Some(&gate));
server.launch().await;
let resumed = wait_for_worker(&run_id);
assert_ne!(resumed, worker);
wait_until_gate_is_polled(&gate);
std::fs::write(&gate, "go").expect("the gate opens");
wait_for_success(&server, &run_id).await;
let (path, commits) = server.workspace_commits(&run_id);
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
assert_eq!(
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
"run\n",
"the crashed attempt's partial output was reset before the rerun; server log:\n{}",
server.stderr_text()
);
let checkpoints = server.checkpoints(&run_id).await;
assert_eq!(checkpoints.len(), 5, "{checkpoints:?}");
let keys: Vec<Option<CheckpointKey>> = checkpoints.iter().map(|(key, _)| Some(*key)).collect();
let committed: Vec<Option<CheckpointKey>> = commits.iter().map(|(_, key)| *key).collect();
assert_eq!(keys, committed);
server.shutdown();
}
/// A worker killed in `prepare_result` before the commit lands: the finish
/// is not durable, the stage reruns once, and one commit exists for it.
#[tokio::test(flavor = "multi_thread")]
async fn a_crash_before_the_commit_lands_reruns_the_stage_once() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let gate = context.temp_dir.join("two.gate");
std::fs::write(&gate, "open").expect("the script gate is open from the start");
let workspace = three_stage_bundle(&context, &gate);
server.hold("commit", "two");
let run_id = run_detached(&context, &server, &workspace);
wait_for_status(&server, &run_id, &["running"]).await;
let worker = wait_for_worker(&run_id);
server.wait_until_held(&run_id, "commit", "two");
crash(&mut server, worker, None);
server.release("commit", "two");
server.launch().await;
wait_for_success(&server, &run_id).await;
let (path, commits) = server.workspace_commits(&run_id);
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
assert_eq!(
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
"run\n",
"the stage reran once, on the snapshot before it"
);
let store = server.petri_store().await;
let outcome = engine::outcome_of(&store, &run_id)
.await
.expect("the run's Petri record inspects");
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
assert!(outcome.complete, "{:?}", outcome.incomplete);
server.shutdown();
}
/// A worker killed after the commit and its durable finish but before the
/// platform record: the restart reconciles the record from the snapshot
/// repository, the stage does not rerun, and one commit exists for it.
#[tokio::test(flavor = "multi_thread")]
async fn a_crash_before_the_record_reconciles_it_from_the_run_branch() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let gate = context.temp_dir.join("two.gate");
std::fs::write(&gate, "open").expect("the script gate is open from the start");
let workspace = three_stage_bundle(&context, &gate);
server.hold("record", "two");
let run_id = run_detached(&context, &server, &workspace);
wait_for_status(&server, &run_id, &["running"]).await;
let worker = wait_for_worker(&run_id);
server.wait_until_held(&run_id, "record", "two");
let before = server.checkpoints(&run_id).await;
assert_eq!(before.len(), 2, "start and one are recorded: {before:?}");
crash(&mut server, worker, None);
server.release("record", "two");
server.launch().await;
wait_for_success(&server, &run_id).await;
let (path, commits) = server.workspace_commits(&run_id);
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
assert_eq!(
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
"run\n",
"the stage with a durable finish did not rerun"
);
let checkpoints = server.checkpoints(&run_id).await;
assert_eq!(checkpoints.len(), 5, "{checkpoints:?}");
let keys: Vec<Option<CheckpointKey>> = checkpoints.iter().map(|(key, _)| Some(*key)).collect();
let committed: Vec<Option<CheckpointKey>> = commits.iter().map(|(_, key)| *key).collect();
assert_eq!(
keys, committed,
"the reconciled record names the one commit"
);
server.shutdown();
}
/// A workspace deleted while the run is down is restored from the snapshot
/// repository, and the next stage sees the checkpoint's files.
#[tokio::test(flavor = "multi_thread")]
async fn a_deleted_workspace_is_restored_from_its_snapshot() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let gate = context.temp_dir.join("two.gate");
let workspace = three_stage_bundle(&context, &gate);
let run_id = run_detached(&context, &server, &workspace);
wait_for_status(&server, &run_id, &["running"]).await;
let worker = wait_for_worker(&run_id);
wait_until_gate_is_polled(&gate);
crash(&mut server, worker, Some(&gate));
let (path, commits) = server.workspace_commits(&run_id);
assert_eq!(subjects(&commits), three_stage_subjects(&run_id)[..2]);
std::fs::remove_dir_all(&path).expect("the workspace is deleted");
server.launch().await;
wait_until_gate_is_polled(&gate);
std::fs::write(&gate, "go").expect("the gate opens");
wait_for_success(&server, &run_id).await;
let (restored, commits) = server.workspace_commits(&run_id);
assert_eq!(restored, path);
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
assert_eq!(
std::fs::read_to_string(restored.join("one.txt")).expect("one.txt was restored"),
"one\n"
);
server.shutdown();
}
/// A stage that fails on its own terms routes to its failure edge on the
/// committed files, and after a crash once the failure is durable the
/// route reruns on the same files.
#[tokio::test(flavor = "multi_thread")]
async fn a_failure_route_sees_the_same_committed_files_after_a_crash() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let gate = context.temp_dir.join("fix.gate");
let workspace = write_petri_workflow(
&context,
&format!(
"digraph Failure {{\n graph [goal=\"Route on failure\", default_max_retries=0]\n \
start [shape=Mdiamond]\n exit [shape=Msquare]\n work [shape=parallelogram, \
script=\"echo partial > out.txt; exit 1\"]\n fix [shape=parallelogram, script=\"test \
\\\"$(cat out.txt)\\\" = partial && echo run >> fix.log && {} && echo fixed >> \
out.txt\"]\n start -> work -> exit\n work -> fix [condition=\"outcome=failed\"]\n \
fix -> exit\n}}\n",
wait_for(&gate)
),
);
let run_id = run_detached(&context, &server, &workspace);
wait_for_status(&server, &run_id, &["running"]).await;
let worker = wait_for_worker(&run_id);
wait_until_gate_is_polled(&gate);
// The route is running on the committed failure: the crash lands here.
crash(&mut server, worker, Some(&gate));
server.launch().await;
wait_until_gate_is_polled(&gate);
std::fs::write(&gate, "go").expect("the gate opens");
wait_for_success(&server, &run_id).await;
let (path, commits) = server.workspace_commits(&run_id);
assert_eq!(subjects(&commits), vec![
format!("fabro({run_id}): start (success)"),
format!("fabro({run_id}): work (failure)"),
format!("fabro({run_id}): fix (success)"),
format!("fabro({run_id}): exit (success)"),
]);
assert_eq!(
std::fs::read_to_string(path.join("out.txt")).expect("out.txt"),
"partial\nfixed\n"
);
assert_eq!(
std::fs::read_to_string(path.join("fix.log")).expect("fix.log"),
"run\n",
"the route saw the failed stage's files, not its own interrupted attempt's"
);
server.shutdown();
}
/// A checkpoint commit that fails ends the run: `checkpoint_failed` is
/// recorded, no route runs, the run is reported failed, and a restart
/// leaves it failed without launching a worker.
#[tokio::test(flavor = "multi_thread")]
async fn a_failed_checkpoint_fails_the_run_and_a_restart_leaves_it_failed() {
if host_plugin().is_none() {
return;
}
let context = test_context!();
let mut server = RunningServer::start().await;
let workspace = write_petri_workflow(
&context,
"digraph Wreck {\n graph [goal=\"Wreck the repository\", default_max_retries=0]\n start \
[shape=Mdiamond]\n exit [shape=Msquare]\n wreck [shape=parallelogram, script=\"rm -rf \
.git && echo garbage > .git\"]\n next [shape=parallelogram, script=\"echo next > \
next.txt\"]\n fix [shape=parallelogram, script=\"echo fix > fix.txt\"]\n start -> wreck \
-> next -> exit\n wreck -> fix [condition=\"outcome=failed\"]\n fix -> exit\n}\n",
);
let run_id = run_detached(&context, &server, &workspace);
let status = wait_for_status(&server, &run_id, &["succeeded", "failed"]).await;
let run = run_json(&server, &format!("runs/{run_id}")).await;
assert_eq!(status, "failed", "run: {run}");
let failures: Vec<String> = settled_events(&server, &run_id, "run.failed")
.await
.iter()
.filter_map(|envelope| match &envelope.event.body {
EventBody::RunFailed(props) => Some(props.failure.detail.message.clone()),
_ => None,
})
.collect();
assert_eq!(failures.len(), 1, "{failures:?}");
assert!(
failures[0].contains("checkpoint commit of `wreck` failed"),
"{failures:?}"
);
let scopes = server.petri_run_dir(&run_id).join("scopes");
let work = std::fs::read_dir(&scopes)
.expect("the scopes directory lists")
.map(|entry| entry.expect("an entry reads").path().join("work"))
.next()
.expect("one workspace");
assert!(!work.join("next.txt").exists(), "no route ran");
assert!(!work.join("fix.txt").exists(), "no route ran");
let store = server.petri_store().await;
let outcome = engine::outcome_of(&store, &run_id)
.await
.expect("the run's Petri record inspects");
assert_ne!(outcome.status, RunStatus::Success, "{outcome:?}");
let checkpoints = server.checkpoints(&run_id).await;
assert_eq!(
checkpoints.len(),
1,
"only start was recorded: {checkpoints:?}"
);
// The restart finds the run terminal and launches nothing for it.
server.kill();
server.launch().await;
assert_eq!(run_status(&server, &run_id).await, "failed");
std::thread::sleep(Duration::from_secs(1));
assert_eq!(
worker_pid(&run_id),
None,
"no worker was launched for the failed run"
);
server.shutdown();
}

View file

@ -18,11 +18,13 @@ use std::sync::Arc;
use axum::extract::DefaultBodyLimit;
use axum::routing::{get, post};
use fabro_api::types::{
PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriOpenResponse, PetriRecord,
PetriRecordList, PetriReleaseRequest, WriteBlobResponse,
PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriOpenResponse, PetriPlatformRecord,
PetriPlatformRecordAppendRequest, PetriPlatformRecordList, PetriRecord, PetriRecordList,
PetriReleaseRequest, WriteBlobResponse,
};
use fabro_petri::petri::{Access, Digest, OwnerId, Record, StoreError};
use fabro_petri::run_store::{log_id_text, parse_log_id};
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord};
use fabro_types::BlobHash;
use fabro_util::error::collect_chain;
use serde_json::{Map, Value, json};
@ -51,6 +53,10 @@ pub(super) fn routes() -> Router<Arc<AppState>> {
post(write_blob).layer(DefaultBodyLimit::disable()),
)
.route("/runs/{id}/petri/blobs/{blobHash}", get(read_blob))
.route(
"/runs/{id}/petri/platform-records",
get(list_platform_records).post(append_platform_record),
)
}
#[derive(serde::Deserialize)]
@ -58,6 +64,11 @@ struct OwnerQuery {
owner: String,
}
#[derive(serde::Deserialize)]
struct KindQuery {
kind: Option<String>,
}
async fn open_run(
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
State(state): State<Arc<AppState>>,
@ -211,6 +222,115 @@ async fn read_blob(
}
}
/// The run's platform records, of one kind when the query names it.
async fn list_platform_records(
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
State(state): State<Arc<AppState>>,
Query(query): Query<KindQuery>,
) -> Response {
let store = state.stores.run_summaries.platform_records();
let records = match query.kind.as_deref() {
Some(kind) => match kind.parse::<PlatformRecordKind>() {
Ok(kind) => store.read_kind(&id, kind).await,
Err(_) => {
return ApiError::bad_request(format!("`{kind}` is not a platform record kind."))
.into_response();
}
},
None => store.read(&id).await,
};
match records {
Ok(records) => match records
.into_iter()
.map(|stored| wire_platform_record(&stored))
.collect::<Result<Vec<_>, _>>()
{
Ok(records) => Json(PetriPlatformRecordList { records }).into_response(),
Err(err) => err.into_response(),
},
Err(err) => platform_store_error_response(id, &err),
}
}
/// Store one platform record for the run and wake its projector.
async fn append_platform_record(
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
State(state): State<Arc<AppState>>,
Json(request): Json<PetriPlatformRecordAppendRequest>,
) -> Response {
let record: PlatformRecord = match serde_json::from_value(Value::Object(request.record)) {
Ok(record) => record,
Err(err) => {
return ApiError::bad_request(format!("Invalid platform record: {err}"))
.into_response();
}
};
let position = match (request.execution, request.firing) {
(Some(execution), Some(firing)) => Some(StagePosition { execution, firing }),
_ => None,
};
let summaries = &state.stores.run_summaries;
match summaries
.platform_records()
.append(&id, &record, position)
.await
{
Ok(stored) => {
summaries.notify_platform_record(id);
match wire_platform_record(&stored) {
Ok(record) => Json(record).into_response(),
Err(err) => err.into_response(),
}
}
Err(err) => platform_store_error_response(id, &err),
}
}
/// A stored platform record as the wire carries it.
fn wire_platform_record(stored: &StoredPlatformRecord) -> Result<PetriPlatformRecord, ApiError> {
let record = serde_json::to_value(&stored.record).map_err(|err| {
ApiError::with_code(
StatusCode::INTERNAL_SERVER_ERROR,
format!(
"The stored platform record at seq {} does not encode: {err}",
stored.seq
),
"petri_store_failed",
)
})?;
let Value::Object(record) = record else {
return Err(ApiError::with_code(
StatusCode::INTERNAL_SERVER_ERROR,
format!(
"The stored platform record at seq {} is not a JSON object.",
stored.seq
),
"petri_store_failed",
));
};
Ok(PetriPlatformRecord {
seq: stored.seq,
recorded_at: stored.recorded_at,
record,
execution: stored.position.map(|position| position.execution),
firing: stored.position.map(|position| position.firing),
})
}
fn platform_store_error_response(run_id: RunId, err: &fabro_store::Error) -> Response {
tracing::error!(
run_id = %run_id,
error = %collect_chain(err).join(": "),
"platform record store failed"
);
ApiError::with_code(
StatusCode::INTERNAL_SERVER_ERROR,
"The platform record store failed; see the server log.",
"petri_store_failed",
)
.into_response()
}
fn unknown_log(log: &str) -> Response {
ApiError::bad_request(format!(
"`{log}` is not a Petri log: expected `coordinator`, `resources` or `execution <n>`."

View file

@ -23,7 +23,10 @@
//! item that follows.
//!
//! After a server restart, [`reconcile_on_startup`] hands a Petri run the
//! previous server left in flight back to a worker in resume mode.
//! previous server left in flight back to a worker in resume mode, once the
//! recovery protocol (`fabro_petri::recovery`) has brought every live
//! workspace to the snapshot its durable state names, or reports the run
//! failed when it cannot.
use std::collections::{BTreeMap, HashSet};
use std::sync::Arc;
@ -34,8 +37,11 @@ use fabro_interview::ControlInterviewer;
use fabro_llm::selection;
use fabro_petri::check::{self, Bundle, CheckError, CheckRequest, Diagnostic, Launch};
use fabro_petri::engine::{self, Conclusion, Execution, RunRequest};
use fabro_petri::hooks::HooksSpec;
use fabro_petri::interview::{Approval, DatabaseQuestions, FabroInterviewer};
use fabro_petri::petri::StoreError;
use fabro_petri::platform_records::SqlitePlatformRecords;
use fabro_petri::recovery::{self, Recovery, RecoveryRequest};
use fabro_petri::runtime::{self, RuntimeSpec};
use fabro_petri::secrets::VaultSecrets;
use fabro_petri::{SqliteRunStore, admission};
@ -378,6 +384,12 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
let observers = vec![petri_interviewer.observer()];
let (_, eligible) = state.resolve_llm_client_with_ready_ids().await;
let dry_run = run_state.spec.settings.run.execution.mode == RunMode::DryRun;
let hooks = HooksSpec::for_run(
Arc::new(SqlitePlatformRecords::new(Arc::clone(
&state.stores.run_summaries,
))),
&run_state.spec.settings.run,
);
let request = RunRequest {
run_id: run_id.to_string(),
run_dir: run_dir.join("petri"),
@ -392,6 +404,7 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
observers,
secrets: Some(Arc::new(VaultSecrets::from_vault(&vault))),
blobs: Some(state.store_ref().blobs()),
hooks: Some(hooks),
};
let result = Box::pin(engine::run(request)).await;
let timing = RunTiming {
@ -430,21 +443,22 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
}
/// Bring a Petri run the server left in flight back to its worker after a
/// restart: the run continues from its records, as Petri's own resume does.
/// restart: the run continues from its records, as Petri's own resume does,
/// on workspaces that match them.
///
/// The lease the previous worker held is released from outside, which
/// fences that worker should it still be alive; then the run is asked to
/// start again as a resume (`run.start_requested` with `resume`, then
/// `run.runnable`, the same pair the API's resume appends), and a managed
/// run is registered for the scheduler in resume mode when Petri's store
/// holds the run, else in start mode: a worker that died before it created
/// the run's record left nothing to continue from, so the run starts from
/// its admitted graphs.
///
/// Full recovery, where the workspace a resumed stage sees is restored to
/// the snapshot its durable state names, is the integration plan's F3.5.
/// Until it lands, a retained workspace is used as the previous worker left
/// it.
/// fences that worker should it still be alive. Then the recovery protocol
/// reads the run's durable execution state: a run with a failed checkpoint
/// is reported failed here and never resumed; otherwise every live
/// workspace on this host is verified against, reset to, or restored from
/// the snapshot its last durable finish names, and a finish with no
/// snapshot fails the run rather than resume it on stale files. The run is
/// then asked to start again as a resume (`run.start_requested` with
/// `resume`, then `run.runnable`, the same pair the API's resume appends),
/// and a managed run is registered for the scheduler in resume mode when
/// Petri's store holds the run, else in start mode: a worker that died
/// before it created the run's record left nothing to continue from, so the
/// run starts from its admitted graphs.
pub(crate) async fn reconcile_on_startup(
state: &Arc<AppState>,
run_id: RunId,
@ -459,8 +473,46 @@ pub(crate) async fn reconcile_on_startup(
return Err(anyhow::Error::new(err).context("releasing the Petri run's lease"));
}
};
let run_dir = Storage::new(state.server_storage_dir())
.run_scratch(&run_id)
.root()
.to_path_buf();
let mode = if held {
RunExecutionMode::Resume
let request = RecoveryRequest::for_run(
run_id,
run_dir.join("petri"),
Arc::new(SqliteRunStore::new(state.db_pool.clone())),
Arc::new(SqlitePlatformRecords::new(Arc::clone(
&state.stores.run_summaries,
))),
&run_state.spec.settings.run,
);
match recovery::recover(request)
.await
.map_err(|err| anyhow::Error::new(err).context("recovering the Petri run"))?
{
Recovery::Start => RunExecutionMode::Start,
Recovery::Resume { workspaces } => {
info!(
run_id = %run_id,
workspaces = workspaces.len(),
"Petri run's workspaces match its durable state"
);
RunExecutionMode::Resume
}
Recovery::Failed { reason } => {
warn!(
run_id = %run_id,
petri_key = %key,
error = %reason,
"Petri run left in flight by the previous server cannot resume; reporting it failed"
);
let (_, _, event) =
failed(FailureReason::WorkflowError, reason, RunTiming::default());
workflow_event::append_event(run_store, &run_id, &event).await?;
return Ok(());
}
}
} else {
RunExecutionMode::Start
};
@ -482,10 +534,6 @@ pub(crate) async fn reconcile_on_startup(
] {
workflow_event::append_event(run_store, &run_id, &event).await?;
}
let run_dir = Storage::new(state.server_storage_dir())
.run_scratch(&run_id)
.root()
.to_path_buf();
let mut runs = state.runs.lock().expect("runs lock poisoned");
runs.insert(
run_id,

View file

@ -62,6 +62,9 @@ const WORKER_ENV_ALLOWLIST: &[&str] = &[
EnvVars::PETRI_SANDBOX_PLUGIN_DEV,
EnvVars::PETRI_SANDBOX_DOCKER_HOST_ADDRESS,
EnvVars::PETRI_SANDBOX_ACTION_HOST_IMAGE,
// A test's checkpoint gates: the worker's hooks hold at a named point
// until the test releases them, so a crash can be placed there.
EnvVars::FABRO_TEST_CHECKPOINT_GATES,
];
const RENDER_GRAPH_ENV_ALLOWLIST: &[&str] = &[EnvVars::PATH, EnvVars::HOME, EnvVars::TMPDIR];

View file

@ -346,3 +346,91 @@ async fn a_worker_store_leases_for_its_launch_not_for_petris_owner() {
drop(created);
wait_until_released(server_store, &key).await;
}
/// A worker stores Fabro's platform records for its run over the API and
/// reads them back by kind: what the checkpoint hooks do from the worker
/// process.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn a_worker_appends_and_reads_platform_records_over_the_api() {
use fabro_petri::platform_records::{HttpPlatformRecords, PlatformRecords};
use fabro_store::platform_records::{CheckpointRecord, DecisionRef, OperationKey};
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
let state = test_app_state();
let base_url = serve(Arc::clone(&state), |router| router).await;
let run_id = RunId::new();
let token = state.test_issue_worker_token(&run_id);
let records =
HttpPlatformRecords::new(worker_client(&base_url, &token, Duration::from_secs(5)).await);
let checkpoint = PlatformRecord::Checkpoint(CheckpointRecord {
execution: 0,
firing: 3,
attempt: Some(1),
workspace: Some("invocation-0-scope-0".to_string()),
git_commit_sha: Some("abc123".to_string()),
diff_summary: None,
patch_blob: None,
operation: Some(OperationKey {
execution: 0,
decision: DecisionRef::AttemptStart {
firing: 3,
attempt: 1,
},
effect: "checkpoint".to_string(),
}),
});
let position = Some(StagePosition {
execution: 0,
firing: 3,
});
let stored = records
.append(&run_id, &checkpoint, position)
.await
.expect("the record appends over the API");
assert_eq!(stored.seq, 1);
assert_eq!(stored.position, position);
let notice = PlatformRecord::RunArchived;
records
.append(&run_id, &notice, None)
.await
.expect("a second record appends");
let checkpoints = records
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
.await
.expect("the checkpoints read back");
assert_eq!(checkpoints.len(), 1);
assert_eq!(checkpoints[0].seq, 1);
assert_eq!(checkpoints[0].position, position);
assert!(matches!(
&checkpoints[0].record,
PlatformRecord::Checkpoint(record) if record.git_commit_sha.as_deref() == Some("abc123")
&& record.operation == checkpoint_operation(&checkpoint)
));
let archived = records
.read_kind(&run_id, PlatformRecordKind::RunArchived)
.await
.expect("the archive record reads back");
assert_eq!(archived.len(), 1);
assert_eq!(archived[0].seq, 2);
assert_eq!(archived[0].position, None);
// Another run's token cannot read this run's records.
let other = state.test_issue_worker_token(&RunId::new());
let foreign =
HttpPlatformRecords::new(worker_client(&base_url, &other, Duration::from_secs(5)).await);
assert!(
foreign
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
.await
.is_err(),
"a worker token names one run"
);
}
fn checkpoint_operation(
record: &fabro_store::PlatformRecord,
) -> Option<fabro_store::platform_records::OperationKey> {
record.operation().cloned()
}

View file

@ -28,6 +28,7 @@ fabro-store = { path = "../fabro-store" }
fabro-types = { path = "../../foundation/fabro-types" }
fabro-vault = { path = "../../foundation/fabro-vault" }
fabro-workflow = { path = "../fabro-workflow" }
fabro-checkpoint = { path = "../fabro-checkpoint" }
fabro-util = { path = "../../foundation/fabro-util" }
petri_runtime.workspace = true
petri_execution.workspace = true
@ -50,6 +51,7 @@ tokio-util.workspace = true
tracing.workspace = true
[dev-dependencies]
fabro-petri = { path = ".", features = ["test-support"] }
fabro-auth = { path = "../../foundation/fabro-auth", features = ["test-support"] }
fabro-llm = { path = "../fabro-llm", features = ["test-support"] }
fabro-store = { path = "../fabro-store", features = ["test-support"] }
@ -57,4 +59,5 @@ fabro-test.workspace = true
fabro-types = { path = "../../foundation/fabro-types", features = ["test-support"] }
petri_testkit.workspace = true
tempfile = "3"
lithos-llm = { workspace = true, features = ["runtime"] }
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }

View file

@ -0,0 +1,864 @@
//! Git snapshots of a Petri run's workspaces: the checkpoint commit and
//! what recovery does with it.
//!
//! A Fabro stage's files are committed on the run branch of its workspace
//! before the stage's finish is recorded, so a durable finish implies a
//! durable snapshot (the integration plan's F3.1). The commit message
//! carries the snapshot's identity as trailers, the run key, execution,
//! firing and attempt, so a restart reconciles a missing platform record
//! from the branch alone.
//!
//! # Where the workspace is
//!
//! Petri's host backend keeps a scope's workspace under the run directory
//! at `scopes/<workspace id>/work`, the layout `HostExecutor::workspace_for`
//! names. This module reaches it there and runs `git` on the host, which
//! is where the worker, and the server at recovery, run. A Docker or
//! Daytona workspace lives inside its sandbox, out of reach of this module:
//! the hooks record that no snapshot was taken and recovery resumes such a
//! run on the retained sandbox as it was left.
//!
//! # The snapshot repository
//!
//! Every checkpoint commit is also pushed to a bare repository beside the
//! run's workspaces, `snapshots/<workspace id>.git`, under an immutable ref
//! per checkpoint (`refs/checkpoints/<execution>/<firing>/<attempt>`). A
//! workspace that is gone at recovery is restored from it, and the refs
//! are what recovery reconciles a missing record from.
use std::path::{Path, PathBuf};
use std::process::Stdio;
use std::time::Duration;
use fabro_checkpoint::author::GitAuthor;
use fabro_checkpoint::trailer::{self, Trailer};
use fabro_store::platform_records::{DecisionRef, OperationKey};
use fabro_types::settings::run::RunCheckpointSettings;
use tokio::process::Command;
use tokio::{fs, time};
/// The failure class of a stage whose checkpoint commit failed: fatal to
/// the run, and terminal for a restart.
pub const CHECKPOINT_FAILED_CLASS: &str = "checkpoint_failed";
/// The effect kind of a checkpoint in its operation identity.
pub const CHECKPOINT_EFFECT: &str = "checkpoint";
pub const RUN_TRAILER: &str = "Fabro-Run";
pub const EXECUTION_TRAILER: &str = "Fabro-Execution";
pub const FIRING_TRAILER: &str = "Fabro-Firing";
pub const ATTEMPT_TRAILER: &str = "Fabro-Attempt";
const FOOTER: &str = "\u{2692}\u{fe0f} Generated with [Fabro](https://fabro.sh)";
const REFS_PREFIX: &str = "refs/checkpoints/";
/// Directories never committed, the legacy executor's list: build output
/// and dependency caches a stage regenerates.
pub const EXCLUDE_DIRS: &[&str] = &[
".git",
"node_modules",
".pnpm-store",
".npm",
"target",
".next",
"__pycache__",
".venv",
"venv",
".cache",
".tox",
".pytest_cache",
];
/// The identity of one snapshot: the attempt whose files it holds.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)]
pub struct CheckpointKey {
pub execution: u64,
pub firing: u64,
pub attempt: u32,
}
impl CheckpointKey {
/// The immutable ref the snapshot is published under.
#[must_use]
pub fn snapshot_ref(self) -> String {
format!(
"{REFS_PREFIX}{}/{}/{}",
self.execution, self.firing, self.attempt
)
}
/// The operation identity of the checkpoint effect: the attempt's
/// decision in its execution, effect kind `checkpoint`.
#[must_use]
pub fn operation(self) -> OperationKey {
OperationKey {
execution: self.execution,
decision: DecisionRef::AttemptStart {
firing: self.firing,
attempt: self.attempt,
},
effect: CHECKPOINT_EFFECT.to_string(),
}
}
/// The key an operation identity names, when it is a checkpoint's.
#[must_use]
pub fn from_operation(operation: &OperationKey) -> Option<Self> {
match operation.decision {
DecisionRef::AttemptStart { firing, attempt }
if operation.effect == CHECKPOINT_EFFECT =>
{
Some(Self {
execution: operation.execution,
firing,
attempt,
})
}
DecisionRef::AttemptStart { .. }
| DecisionRef::ExecutionStart
| DecisionRef::Route { .. } => None,
}
}
fn from_ref(name: &str) -> Option<Self> {
let mut parts = name.strip_prefix(REFS_PREFIX)?.split('/');
let execution = parts.next()?.parse().ok()?;
let firing = parts.next()?.parse().ok()?;
let attempt = parts.next()?.parse().ok()?;
parts.next().is_none().then_some(Self {
execution,
firing,
attempt,
})
}
/// The key a checkpoint commit's message carries in its trailers.
#[must_use]
pub fn from_message(message: &str) -> Option<Self> {
Some(Self {
execution: trailer::parse(message, EXECUTION_TRAILER)?.parse().ok()?,
firing: trailer::parse(message, FIRING_TRAILER)?.parse().ok()?,
attempt: trailer::parse(message, ATTEMPT_TRAILER)?.parse().ok()?,
})
}
}
/// Why a snapshot could not be taken, found or restored.
#[derive(Debug, thiserror::Error)]
pub enum CheckpointError {
#[error("the workspace `{workspace}` does not exist at {}", path.display())]
WorkspaceMissing {
workspace: String,
path: PathBuf,
},
#[error("git {action} failed ({status}): {detail}")]
Command {
action: String,
status: String,
detail: String,
},
#[error("git {action} could not run")]
Spawn {
action: String,
#[source]
source: std::io::Error,
},
#[error("git {action} did not finish within {timeout:?}")]
TimedOut { action: String, timeout: Duration },
#[error("the workspace could not be prepared at {}", path.display())]
Io {
path: PathBuf,
#[source]
source: std::io::Error,
},
#[error("the restored workspace is at {actual}, not the snapshot {expected}")]
RestoreMismatch { expected: String, actual: String },
}
/// A checkpoint commit: the commit, and whether an earlier attempt of the
/// same operation had already made it.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct Snapshot {
pub sha: String,
pub reused: bool,
}
/// One published snapshot of a workspace.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct PublishedSnapshot {
pub key: CheckpointKey,
pub sha: String,
}
/// The workspaces of one run on this host, and the Git operations Fabro
/// performs on them.
#[derive(Clone, Debug)]
pub struct RunWorkspaces {
run_dir: PathBuf,
run_id: String,
author: GitAuthor,
exclude_globs: Vec<String>,
timeout: Duration,
}
impl RunWorkspaces {
#[must_use]
pub fn new(
run_dir: PathBuf,
run_id: String,
author: GitAuthor,
settings: &RunCheckpointSettings,
) -> Self {
Self {
run_dir,
run_id,
author,
exclude_globs: settings.exclude_globs.clone(),
timeout: Duration::from_millis(settings.commit_timeout_ms.max(1)),
}
}
/// The run branch every workspace of the run commits on.
#[must_use]
pub fn run_branch(&self) -> String {
format!("fabro/run/{}", self.run_id)
}
/// Where the host backend keeps the workspace: `scopes/<id>/work` under
/// the run directory.
#[must_use]
pub fn workspace_path(&self, workspace: &str) -> PathBuf {
self.run_dir.join("scopes").join(workspace).join("work")
}
/// The bare repository the workspace's snapshots are published to.
#[must_use]
pub fn snapshot_repository(&self, workspace: &str) -> PathBuf {
self.run_dir
.join("snapshots")
.join(format!("{workspace}.git"))
}
/// Whether the workspace exists on this host.
pub async fn workspace_exists(&self, workspace: &str) -> bool {
fs::try_exists(self.workspace_path(workspace))
.await
.unwrap_or(false)
}
/// Commit the workspace's files on the run branch as the snapshot of
/// `key`, and publish it. An earlier commit of the same key that the
/// workspace still sits on, unchanged, is reused.
pub async fn commit(
&self,
workspace: &str,
key: CheckpointKey,
node: &str,
status: &str,
) -> Result<Snapshot, CheckpointError> {
let path = self.workspace_path(workspace);
if !self.workspace_exists(workspace).await {
return Err(CheckpointError::WorkspaceMissing {
workspace: workspace.to_string(),
path,
});
}
self.ensure_repository(&path).await?;
if let Some(existing) = self.published_sha(workspace, key).await? {
if self.head(&path).await?.as_deref() == Some(existing.as_str())
&& self.is_clean(&path).await?
{
return Ok(Snapshot {
sha: existing,
reused: true,
});
}
}
let mut add = vec![
"add".to_string(),
"-A".to_string(),
"--".to_string(),
".".to_string(),
];
add.extend(
EXCLUDE_DIRS
.iter()
.map(|dir| format!(":(glob,exclude)**/{dir}/**")),
);
add.extend(
self.exclude_globs
.iter()
.map(|glob| format!(":(glob,exclude){glob}")),
);
self.git(&path, "add", &add).await?;
let message = self.message(key, node, status);
let user_name = format!("user.name={}", self.author.name);
let user_email = format!("user.email={}", self.author.email);
self.git(&path, "commit", &[
"-c",
&user_name,
"-c",
&user_email,
"commit",
"-q",
"--allow-empty",
"-m",
&message,
])
.await?;
let sha = self.git(&path, "rev-parse", &["rev-parse", "HEAD"]).await?;
self.publish(workspace, &path, key, &sha).await?;
Ok(Snapshot { sha, reused: false })
}
/// The commit of `key`, from the snapshot repository first, else from
/// the workspace's own history by the trailers.
pub async fn find(
&self,
workspace: &str,
key: CheckpointKey,
) -> Result<Option<String>, CheckpointError> {
if let Some(sha) = self.published_sha(workspace, key).await? {
return Ok(Some(sha));
}
let path = self.workspace_path(workspace);
if !self.workspace_exists(workspace).await || self.head(&path).await?.is_none() {
return Ok(None);
}
let listed = self
.git(&path, "log", &[
"log",
"--format=%H",
"--extended-regexp",
&format!("--grep=^{EXECUTION_TRAILER}: {}$", key.execution),
&format!("--grep=^{FIRING_TRAILER}: {}$", key.firing),
&format!("--grep=^{ATTEMPT_TRAILER}: {}$", key.attempt),
"--all-match",
"HEAD",
])
.await?;
Ok(listed.lines().next().map(str::to_owned))
}
/// Every snapshot published for the workspace.
pub async fn published(
&self,
workspace: &str,
) -> Result<Vec<PublishedSnapshot>, CheckpointError> {
let repository = self.snapshot_repository(workspace);
if !fs::try_exists(&repository).await.unwrap_or(false) {
return Ok(Vec::new());
}
let listed = self
.git(&repository, "for-each-ref", &[
"for-each-ref",
"--format=%(refname) %(objectname)",
REFS_PREFIX,
])
.await?;
Ok(listed
.lines()
.filter_map(|line| {
let (name, sha) = line.split_once(' ')?;
Some(PublishedSnapshot {
key: CheckpointKey::from_ref(name)?,
sha: sha.to_string(),
})
})
.collect())
}
/// Whether `ancestor` is reachable from `descendant` in the workspace's
/// published history.
pub async fn is_ancestor(
&self,
workspace: &str,
ancestor: &str,
descendant: &str,
) -> Result<bool, CheckpointError> {
let repository = self.snapshot_repository(workspace);
Ok(self
.git_status(&repository, "merge-base", &[
"merge-base",
"--is-ancestor",
ancestor,
descendant,
])
.await?
.is_some())
}
/// The workspace's `HEAD`, or `None` when it has no commit.
pub async fn workspace_head(&self, workspace: &str) -> Result<Option<String>, CheckpointError> {
let path = self.workspace_path(workspace);
self.head(&path).await
}
/// Whether the workspace sits on `sha` with nothing changed since.
pub async fn matches(&self, workspace: &str, sha: &str) -> Result<bool, CheckpointError> {
let path = self.workspace_path(workspace);
Ok(self.head(&path).await?.as_deref() == Some(sha) && self.is_clean(&path).await?)
}
/// Bring the workspace back to `sha`: tracked files reset, untracked
/// files removed, the excluded caches left alone.
pub async fn reset(&self, workspace: &str, sha: &str) -> Result<(), CheckpointError> {
let path = self.workspace_path(workspace);
self.git(&path, "reset", &["reset", "-q", "--hard", sha])
.await?;
let mut clean = vec!["clean".to_string(), "-fdq".to_string()];
for dir in EXCLUDE_DIRS {
clean.push("-e".to_string());
clean.push((*dir).to_string());
}
for glob in &self.exclude_globs {
clean.push("-e".to_string());
clean.push(glob.clone());
}
self.git(&path, "clean", &clean).await?;
Ok(())
}
/// Recreate a gone workspace from the published snapshot `key`, at
/// `sha`, on the run branch.
pub async fn restore(
&self,
workspace: &str,
key: CheckpointKey,
sha: &str,
) -> Result<(), CheckpointError> {
let path = self.workspace_path(workspace);
fs::create_dir_all(&path)
.await
.map_err(|source| CheckpointError::Io {
path: path.clone(),
source,
})?;
self.git(&path, "init", &["init", "-q"]).await?;
let repository = self.snapshot_repository(workspace);
let repository = repository.to_string_lossy().into_owned();
self.git(&path, "fetch", &[
"fetch",
"-q",
&repository,
&key.snapshot_ref(),
])
.await?;
let branch = self.run_branch();
self.git(&path, "checkout", &[
"checkout",
"-q",
"-B",
&branch,
"FETCH_HEAD",
])
.await?;
let actual = self.git(&path, "rev-parse", &["rev-parse", "HEAD"]).await?;
if actual != sha {
return Err(CheckpointError::RestoreMismatch {
expected: sha.to_string(),
actual,
});
}
Ok(())
}
/// The commit message: Fabro's subject, the footer, and the identity
/// trailers last, so `git interpret-trailers` and
/// [`CheckpointKey::from_message`] both read them.
fn message(&self, key: CheckpointKey, node: &str, status: &str) -> String {
let subject = format!("fabro({}): {node} ({status})", self.run_id);
let execution = key.execution.to_string();
let firing = key.firing.to_string();
let attempt = key.attempt.to_string();
let mut trailers = vec![
Trailer {
key: RUN_TRAILER,
value: &self.run_id,
},
Trailer {
key: EXECUTION_TRAILER,
value: &execution,
},
Trailer {
key: FIRING_TRAILER,
value: &firing,
},
Trailer {
key: ATTEMPT_TRAILER,
value: &attempt,
},
];
let defaults = GitAuthor::default();
let co_author = format!("{} <{}>", defaults.name, defaults.email);
if !self.author.is_default() {
trailers.push(Trailer {
key: "Co-Authored-By",
value: &co_author,
});
}
trailer::format_message(&subject, FOOTER, &trailers)
}
/// A repository on the run branch, initialised when the workspace has
/// none.
async fn ensure_repository(&self, path: &Path) -> Result<(), CheckpointError> {
if self
.git_status(path, "rev-parse", &["rev-parse", "--git-dir"])
.await?
.is_none()
{
self.git(path, "init", &["init", "-q"]).await?;
}
let branch = self.run_branch();
let current = self
.git_status(path, "symbolic-ref", &[
"symbolic-ref",
"-q",
"--short",
"HEAD",
])
.await?;
if current.as_deref() != Some(branch.as_str()) {
self.git(path, "checkout", &["checkout", "-q", "-B", &branch])
.await?;
}
Ok(())
}
async fn publish(
&self,
workspace: &str,
path: &Path,
key: CheckpointKey,
sha: &str,
) -> Result<(), CheckpointError> {
let repository = self.snapshot_repository(workspace);
if !fs::try_exists(&repository).await.unwrap_or(false) {
fs::create_dir_all(&repository)
.await
.map_err(|source| CheckpointError::Io {
path: repository.clone(),
source,
})?;
self.git(&repository, "init --bare", &["init", "-q", "--bare"])
.await?;
}
let refspec = format!("{sha}:{}", key.snapshot_ref());
let repository = repository.to_string_lossy().into_owned();
self.git(path, "push", &[
"push",
"-q",
"--force",
&repository,
&refspec,
])
.await?;
Ok(())
}
async fn published_sha(
&self,
workspace: &str,
key: CheckpointKey,
) -> Result<Option<String>, CheckpointError> {
let repository = self.snapshot_repository(workspace);
if !fs::try_exists(&repository).await.unwrap_or(false) {
return Ok(None);
}
self.git_status(&repository, "rev-parse", &[
"rev-parse",
"-q",
"--verify",
&key.snapshot_ref(),
])
.await
}
async fn head(&self, path: &Path) -> Result<Option<String>, CheckpointError> {
self.git_status(path, "rev-parse", &["rev-parse", "-q", "--verify", "HEAD"])
.await
}
async fn is_clean(&self, path: &Path) -> Result<bool, CheckpointError> {
let status = self.git(path, "status", &["status", "--porcelain"]).await?;
Ok(status.trim().is_empty())
}
/// Run `git` in `cwd`; a non-zero exit is the error.
async fn git<S: AsRef<str>>(
&self,
cwd: &Path,
action: &str,
args: &[S],
) -> Result<String, CheckpointError> {
let output = self.run(cwd, action, args).await?;
if output.status.success() {
Ok(String::from_utf8_lossy(&output.stdout).trim().to_owned())
} else {
Err(CheckpointError::Command {
action: action.to_string(),
status: output.status.to_string(),
detail: detail(&output.stderr),
})
}
}
/// Run `git` in `cwd`; a non-zero exit is `None`, for the queries whose
/// answer it is (an unborn `HEAD`, a missing ref, no repository).
async fn git_status<S: AsRef<str>>(
&self,
cwd: &Path,
action: &str,
args: &[S],
) -> Result<Option<String>, CheckpointError> {
let output = self.run(cwd, action, args).await?;
Ok(output
.status
.success()
.then(|| String::from_utf8_lossy(&output.stdout).trim().to_owned()))
}
async fn run<S: AsRef<str>>(
&self,
cwd: &Path,
action: &str,
args: &[S],
) -> Result<std::process::Output, CheckpointError> {
let mut command = Command::new("git");
command
.args([
"-c",
"core.hooksPath=/dev/null",
"-c",
"commit.gpgsign=false",
"-c",
"gc.auto=0",
"-c",
"advice.detachedHead=false",
"-c",
"init.defaultBranch=main",
])
.args(args.iter().map(AsRef::as_ref))
.current_dir(cwd)
.env("GIT_TERMINAL_PROMPT", "0")
.env_remove("GIT_DIR")
.env_remove("GIT_WORK_TREE")
.env_remove("GIT_INDEX_FILE")
.stdin(Stdio::null())
.stdout(Stdio::piped())
.stderr(Stdio::piped())
.kill_on_drop(true);
match time::timeout(self.timeout, command.output()).await {
Ok(Ok(output)) => Ok(output),
Ok(Err(source)) => Err(CheckpointError::Spawn {
action: action.to_string(),
source,
}),
Err(_) => Err(CheckpointError::TimedOut {
action: action.to_string(),
timeout: self.timeout,
}),
}
}
}
/// The tail of git's stderr for an error message: what the run's record
/// carries about the failure, bounded.
fn detail(stderr: &[u8]) -> String {
const LIMIT: usize = 512;
let text = String::from_utf8_lossy(stderr);
let text = text.trim();
if text.is_empty() {
return "no output".to_string();
}
let start = text.len().saturating_sub(LIMIT);
let start = text
.char_indices()
.map(|(index, _)| index)
.find(|index| *index >= start)
.unwrap_or(0);
text[start..].to_string()
}
#[cfg(test)]
mod tests {
use super::*;
fn workspaces(dir: &Path) -> RunWorkspaces {
RunWorkspaces::new(
dir.to_path_buf(),
"run-1".to_string(),
GitAuthor::default(),
&RunCheckpointSettings::default(),
)
}
#[test]
fn a_key_round_trips_through_its_ref_and_its_operation() {
let key = CheckpointKey {
execution: 3,
firing: 17,
attempt: 2,
};
assert_eq!(key.snapshot_ref(), "refs/checkpoints/3/17/2");
assert_eq!(CheckpointKey::from_ref(&key.snapshot_ref()), Some(key));
assert_eq!(CheckpointKey::from_ref("refs/heads/main"), None);
assert_eq!(CheckpointKey::from_operation(&key.operation()), Some(key));
assert_eq!(
CheckpointKey::from_operation(&OperationKey {
execution: 3,
decision: DecisionRef::Route {
firing: 17,
attempt: 2,
},
effect: CHECKPOINT_EFFECT.to_string(),
}),
None
);
}
#[test]
fn the_message_carries_the_identity_as_trailers_last() {
let dir = tempfile::tempdir().expect("a temp dir");
let key = CheckpointKey {
execution: 0,
firing: 4,
attempt: 1,
};
let message = workspaces(dir.path()).message(key, "build", "success");
assert!(message.starts_with("fabro(run-1): build (success)\n\n"));
assert_eq!(CheckpointKey::from_message(&message), Some(key));
assert_eq!(trailer::parse(&message, RUN_TRAILER), Some("run-1"));
}
#[tokio::test]
async fn a_commit_is_published_found_and_restored() {
let dir = tempfile::tempdir().expect("a temp dir");
let workspaces = workspaces(dir.path());
let workspace = "invocation-0-scope-0";
let path = workspaces.workspace_path(workspace);
fs::create_dir_all(&path).await.expect("the workspace");
fs::write(path.join("out.txt"), "one\n")
.await
.expect("a file");
let key = CheckpointKey {
execution: 0,
firing: 2,
attempt: 1,
};
let first = workspaces
.commit(workspace, key, "build", "success")
.await
.expect("the commit");
assert!(!first.reused);
let again = workspaces
.commit(workspace, key, "build", "success")
.await
.expect("the second commit");
assert_eq!(again, Snapshot {
sha: first.sha.clone(),
reused: true,
});
assert_eq!(
workspaces.find(workspace, key).await.expect("the lookup"),
Some(first.sha.clone())
);
assert_eq!(
workspaces.published(workspace).await.expect("the listing"),
vec![PublishedSnapshot {
key,
sha: first.sha.clone(),
}]
);
assert!(
workspaces
.matches(workspace, &first.sha)
.await
.expect("matches")
);
// The stage goes on, then the workspace is lost.
fs::write(path.join("out.txt"), "two\n")
.await
.expect("a change");
fs::write(path.join("scratch.txt"), "junk\n")
.await
.expect("an untracked file");
assert!(
!workspaces
.matches(workspace, &first.sha)
.await
.expect("matches")
);
workspaces
.reset(workspace, &first.sha)
.await
.expect("the reset");
assert_eq!(
fs::read_to_string(path.join("out.txt"))
.await
.expect("the file"),
"one\n"
);
assert!(
!fs::try_exists(path.join("scratch.txt"))
.await
.expect("exists")
);
fs::remove_dir_all(&path).await.expect("the workspace goes");
assert_eq!(
workspaces.find(workspace, key).await.expect("the lookup"),
Some(first.sha.clone()),
"the snapshot repository still knows the commit"
);
workspaces
.restore(workspace, key, &first.sha)
.await
.expect("the restore");
assert_eq!(
fs::read_to_string(path.join("out.txt"))
.await
.expect("the restored file"),
"one\n"
);
assert_eq!(
workspaces
.workspace_head(workspace)
.await
.expect("the head"),
Some(first.sha)
);
}
#[tokio::test]
async fn an_unusable_repository_fails_the_commit() {
let dir = tempfile::tempdir().expect("a temp dir");
let workspaces = workspaces(dir.path());
let workspace = "invocation-0-scope-0";
let path = workspaces.workspace_path(workspace);
fs::create_dir_all(&path).await.expect("the workspace");
fs::write(path.join(".git"), "garbage\n")
.await
.expect("a broken gitfile");
let key = CheckpointKey {
execution: 0,
firing: 2,
attempt: 1,
};
let error = workspaces
.commit(workspace, key, "build", "success")
.await
.expect_err("the commit fails");
assert!(matches!(error, CheckpointError::Command { .. }), "{error}");
}
#[test]
fn detail_keeps_the_tail_of_long_output() {
let long = "x".repeat(600);
assert_eq!(detail(long.as_bytes()).len(), 512);
assert_eq!(detail(b""), "no output");
}
}

View file

@ -18,17 +18,20 @@
//! What the caller supplies beyond the runtime: the interviewer its
//! questions go to ([`interview`](crate::interview) in the worker and the
//! server), the secret provider over the vault ([`secrets`](crate::secrets))
//! and the blob table ([`blobs`](crate::blobs)) when it has them. What the
//! standalone runner's defaults give the run: Petri's local hook service
//! for `[[run.hooks]]`, no `ExecutionHooks` of Fabro's own, no host tools,
//! and `Retention::Always` for every workspace, Fabro's default.
//! and the blob table ([`blobs`](crate::blobs)) when it has them, and a
//! [`HooksSpec`] for Fabro's own [`FabroHooks`], which wrap Petri's local
//! hook service: the checkpoint commit before every durable finish and its
//! platform record after every route, with a failed commit ending the run
//! as a `checkpoint_failed` failure. What the standalone runner's defaults
//! give the run: Petri's local hook service for `[[run.hooks]]`, no host
//! tools, and `Retention::Always` for every workspace, Fabro's default.
//! Cancellation rides the caller's token: when it fires, the root
//! invocation is cancelled politely and Petri records why.
//!
//! A resume here is Petri's own: the run continues from its records, and
//! sandbox leases are reconciled by label. Full recovery, where the
//! workspace a resumed stage sees is restored to the snapshot its durable
//! state names, is the integration plan's F3.5 and lands after this.
//! sandbox leases are reconciled by label. What the workspaces look like
//! when it does is the server's business before it relaunches the worker
//! ([`recovery`](crate::recovery)).
//!
//! No stage or agent event is projected into Fabro's tables here; the
//! caller appends only the run lifecycle events Fabro's read side needs to
@ -38,13 +41,14 @@
use std::path::PathBuf;
use std::sync::Arc;
use fabro_types::{FailureReason, SandboxProviderKind};
use fabro_types::{FailureReason, RunId, SandboxProviderKind};
use petri_execution::host::{self, HostError, HostRun};
use petri_execution::inspect::{self, InspectError, RunInspection};
use petri_execution::{
Access, CancelReason, ExecutionObserver, InterviewDispatcher, Interviewer, InvocationId,
RECEIPT_FILE, RunKey, RunStore,
};
use petri_runtime::driver::lifecycle::ExecutionHooks;
use petri_runtime::executor::{Retention, SecretProvider};
use petri_runtime::{RunOptions, SandboxBackend};
use tokio::fs;
@ -53,6 +57,7 @@ use tracing::{debug, info, warn};
use crate::admission::AdmittedGraphs;
use crate::blobs::{Blobs, RunBlobs};
use crate::hooks::{FabroHooks, HooksSpec};
use crate::runtime::RuntimeSpec;
use crate::secrets::SharedSecrets;
@ -95,6 +100,9 @@ pub struct RunRequest {
/// Where offloaded stage values go; `None` keeps Petri's local store
/// under the run directory.
pub blobs: Option<Arc<dyn Blobs>>,
/// Fabro's hooks: the checkpoint commit and its record. `None` runs
/// with Petri's local hook service alone.
pub hooks: Option<HooksSpec>,
}
/// The recorded status of a finished run.
@ -168,12 +176,32 @@ pub async fn run(request: RunRequest) -> Result<RunOutcome, RunError> {
if let Some(blobs) = request.blobs {
runtime = runtime.capability(RunBlobs::output_store(blobs));
}
let fabro_hooks = request.hooks.map(|spec| {
let inner = runtime
.installed_hooks()
.unwrap_or_else(|| Arc::new(NoHooks));
let run_id = spec_run_id(&request.run_id);
Arc::new(FabroHooks::new(
spec,
inner,
run_id,
key.clone(),
request.run_dir.clone(),
Arc::clone(&request.store),
))
});
if let Some(hooks) = &fabro_hooks {
runtime = runtime.hooks(Arc::clone(hooks) as Arc<dyn ExecutionHooks>);
}
let dispatcher = InterviewDispatcher::new(request.interviewer);
let cancel = request.cancel.clone();
let mut cancel_task = None;
let with_handle = |handle: petri_execution::CoordinatorHandle, secrets| {
dispatcher.wire(handle.clone(), secrets);
if let Some(hooks) = &fabro_hooks {
hooks.attach(handle.clone());
}
cancel_task = Some(tokio::spawn(async move {
cancel.cancelled().await;
info!("cancelling the Petri run");
@ -213,9 +241,37 @@ pub async fn run(request: RunRequest) -> Result<RunOutcome, RunError> {
Err(error) => warn!(error = %error, "Petri run ended with a host error"),
}
let inspection = inspect(request.store.as_ref(), &key).await?;
outcome(inspection, result.err())
let mut outcome = outcome(inspection, result.err())?;
// A failed checkpoint cancelled the run; what Fabro reports is the
// checkpoint failure, not a cancellation.
if let Some(failure) = fabro_hooks
.as_ref()
.and_then(|hooks| hooks.checkpoint_failure())
{
outcome.status = RunStatus::Failed;
outcome.failure = Some(failure);
}
Ok(outcome)
}
/// The Fabro run id the run key names. A key that is not one (a test's
/// bare key) still gets hooks, under a fresh id for its platform records.
fn spec_run_id(run_id: &str) -> RunId {
run_id.parse().unwrap_or_else(|_| {
warn!(
run_id,
"the Petri run key is not a Fabro run id; platform records use a fresh id"
);
RunId::new()
})
}
/// No host hooks at all: what Fabro's hooks wrap when the runtime installed
/// none.
struct NoHooks;
impl ExecutionHooks for NoHooks {}
/// What the run's record says, read through a handle that holds no lease:
/// the same derivation [`run`] ends with, for a caller that only holds the
/// store, such as a test checking a finished run.

View file

@ -0,0 +1,583 @@
//! Fabro's awaited extension points on a Petri run: the checkpoint commit,
//! its platform record, and the run-level ends, wrapped around Petri's own
//! hook service so `[[run.hooks]]` keep running.
//!
//! [`FabroHooks`] implements Petri's `ExecutionHooks` and is installed with
//! `Runtime::hooks` by [`engine::run`](crate::engine::run). It holds the
//! hooks the runtime installed before it (Petri's local hook service behind
//! its adapter, which serves `[[run.hooks]]`) and forwards every point to
//! them, `run_finished` and `scope_released` included, the way Petri's
//! embedding host does. Its own work, at each point:
//!
//! - `prepare_result`: the checkpoint commit, before the `StepFinished` record
//! is appended, so a durable finish implies a durable snapshot. A stage that
//! failed on its own terms is committed like a successful one; only a
//! cancelled attempt is not. A failed commit is fatal to the run: the outcome
//! becomes a failure of class `checkpoint_failed`, the run is cancelled
//! through the coordinator handle, and `transition` refuses the firing's
//! routes, so no route is taken.
//! - `transition`: the platform checkpoint record, keyed on the Petri position
//! and the checkpoint's operation identity. A failed write is a recorded
//! problem on the transition, never a blocked route.
//! - `run_finished` and `scope_released`: forwarded, so the local service runs
//! `run_complete`, `run_failed` and `sandbox_cleanup` with the sandbox in
//! place. Fabro's own end-of-run work (the terminal lifecycle event,
//! notifications on it) is the run lifecycle path's, on the worker's and
//! server's side of the engine, and the workspace's retention is Petri's
//! (`Retention::Always`).
//!
//! # Operation identities
//!
//! Every external effect here is keyed on `(run key, execution, DecisionId,
//! effect kind)` from the hook context and deduplicated on retry: the
//! checkpoint's key is the attempt's decision in its execution, effect
//! `checkpoint`. A re-dispatched attempt whose commit already landed
//! reuses it when the workspace still sits on it unchanged (see
//! [`RunWorkspaces::commit`]); a reissued routing decision finds the
//! record, or the commit by its trailers, and writes nothing twice.
//!
//! # Where the workspace is
//!
//! The commit runs on the host, in the workspace Petri's host backend keeps
//! under the run directory (`crate::checkpoint`). A run on Docker or
//! Daytona has no workspace this process can reach; its hooks record that
//! no snapshot was taken and leave the run to continue as before.
use std::collections::{HashMap, HashSet};
use std::path::PathBuf;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, Mutex, MutexGuard, OnceLock, PoisonError};
use std::time::Duration;
use fabro_checkpoint::author::GitAuthor;
use fabro_store::platform_records::CheckpointRecord;
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace};
use fabro_types::{RunId, SandboxProviderKind};
use fabro_util::error::collect_chain;
use petri_execution::{CancelReason, CoordinatorHandle, InvocationId, RunKey, RunStore};
use petri_runtime::driver::lifecycle::{
AdmitAttempt, AttemptDecision, ExecutionHooks, HookContext, Note, PrepareError, PrepareResult,
Prepared, Recorded, ResultOrigin, RunFinished, ScopeReleased, Transition, TransitionError,
TransitionReport,
};
use petri_runtime::ir::{FailureInfo, ScopeId, Status};
use serde_json::json;
use tokio::sync::{Mutex as AsyncMutex, OnceCell};
use tokio::{fs, time};
use tracing::{debug, info, warn};
use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces};
use crate::platform_records::PlatformRecords;
use crate::workspace::{self, WorkspaceLookup};
/// The note kind the hooks record on a firing about its checkpoint.
pub const CHECKPOINT_NOTE: &str = "fabro.checkpoint";
/// How often a held checkpoint polls its test gate.
const GATE_POLL: Duration = Duration::from_millis(50);
/// What Fabro's hooks need beside the run: where the platform records go,
/// who authors the commits, and the checkpoint settings.
pub struct HooksSpec {
pub records: Arc<dyn PlatformRecords>,
pub author: GitAuthor,
pub checkpoint: RunCheckpointSettings,
/// Whether the run's workspaces are on this host (the local sandbox
/// provider). A run elsewhere takes no snapshot.
pub host_workspaces: bool,
/// A test's gate directory: a checkpoint point named by a `.hold` file
/// there waits for its `.release` file. `None` outside tests.
pub test_gates: Option<PathBuf>,
}
impl HooksSpec {
/// The spec a run's settings give: its Git author, its checkpoint
/// settings, and whether its sandbox provider keeps workspaces on this
/// host.
#[must_use]
pub fn for_run(records: Arc<dyn PlatformRecords>, settings: &RunNamespace) -> Self {
Self {
records,
author: settings
.git
.author
.as_ref()
.map(GitAuthor::from)
.unwrap_or_default(),
checkpoint: settings.checkpoint.clone(),
host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL,
test_gates: None,
}
}
#[must_use]
pub fn with_test_gates(mut self, gates: Option<PathBuf>) -> Self {
self.test_gates = gates;
self
}
}
/// Fabro's `ExecutionHooks`, around the hooks the runtime installed.
pub struct FabroHooks {
inner: Arc<dyn ExecutionHooks>,
run_id: RunId,
records: Arc<dyn PlatformRecords>,
workspaces: RunWorkspaces,
lookup: WorkspaceLookup,
host_workspaces: bool,
test_gates: Option<PathBuf>,
handle: OnceLock<CoordinatorHandle>,
/// The workspace and commit of every checkpoint this process made.
committed: Mutex<HashMap<CheckpointKey, (String, String)>>,
/// Which checkpoints have their platform record, loaded from the store
/// once and kept up to date with every append.
recorded: Mutex<HashSet<CheckpointKey>>,
recorded_loaded: OnceCell<()>,
/// Inherited workspaces resolved through the run's records.
inherited: Mutex<HashMap<InvocationId, Option<String>>>,
/// One lock per workspace: the branches of a parallel node and a nested
/// invocation share their caller's workspace, and Git allows one index
/// operation at a time in it.
workspace_locks: Mutex<HashMap<String, Arc<AsyncMutex<()>>>>,
/// The checkpoint failure that ended the run, when one did.
failure: Mutex<Option<String>>,
unreachable_noted: AtomicBool,
}
impl FabroHooks {
/// Wrap `inner` (the hooks `Runtime::installed_hooks` returned) for the
/// run whose records are in `store` under `run_key`, with its
/// workspaces under `run_dir`.
#[must_use]
pub fn new(
spec: HooksSpec,
inner: Arc<dyn ExecutionHooks>,
run_id: RunId,
run_key: RunKey,
run_dir: PathBuf,
store: Arc<dyn RunStore>,
) -> Self {
let workspaces =
RunWorkspaces::new(run_dir, run_id.to_string(), spec.author, &spec.checkpoint);
Self {
inner,
run_id,
records: spec.records,
workspaces,
lookup: WorkspaceLookup::new(store, run_key),
host_workspaces: spec.host_workspaces,
test_gates: spec.test_gates,
handle: OnceLock::new(),
committed: Mutex::default(),
recorded: Mutex::default(),
recorded_loaded: OnceCell::new(),
inherited: Mutex::default(),
workspace_locks: Mutex::default(),
failure: Mutex::default(),
unreachable_noted: AtomicBool::new(false),
}
}
/// Hand the hooks the running coordinator, so a fatal checkpoint can
/// cancel the run. Called once, from the host's handle callback.
pub fn attach(&self, handle: CoordinatorHandle) {
if self.handle.set(handle).is_err() {
debug!("the coordinator handle was already attached to the hooks");
}
}
/// The checkpoint failure that ended the run, when one did: what the
/// engine reports the run failed with.
#[must_use]
pub fn checkpoint_failure(&self) -> Option<String> {
lock(&self.failure).clone()
}
/// The run's workspaces on this host, as the hooks reach them.
#[must_use]
pub fn workspaces(&self) -> &RunWorkspaces {
&self.workspaces
}
/// The lock that serializes Git work in one workspace.
fn workspace_lock(&self, workspace: &str) -> Arc<AsyncMutex<()>> {
Arc::clone(
lock(&self.workspace_locks)
.entry(workspace.to_string())
.or_default(),
)
}
fn fail_run(&self, message: &str) {
let mut failure = lock(&self.failure);
if failure.is_none() {
*failure = Some(message.to_string());
}
drop(failure);
if let Some(handle) = self.handle.get() {
info!(run_id = %self.run_id, "cancelling the Petri run after a failed checkpoint");
handle.cancel_root_for(CancelReason::Control);
} else {
warn!(
run_id = %self.run_id,
"no coordinator handle is attached; the failed checkpoint cannot cancel the run"
);
}
}
/// The workspace id of `scope` in the context's invocation: the
/// isolated name when its workspace exists, else the inherited one the
/// records name, else the isolated name for the caller to report.
async fn workspace_of(&self, context: &HookContext, scope: ScopeId) -> Result<String, String> {
let isolated = workspace::isolated_workspace(context.invocation, scope);
if self.workspaces.workspace_exists(&isolated).await {
return Ok(isolated);
}
let cached = lock(&self.inherited).get(&context.invocation).cloned();
let inherited = if let Some(inherited) = cached {
inherited
} else {
let inherited = self
.lookup
.inherited(context.invocation)
.await
.map_err(|error| {
format!(
"the workspace of scope {scope} in invocation {} could not be found: {}",
context.invocation,
collect_chain(&error).join(": ")
)
})?;
lock(&self.inherited).insert(context.invocation, inherited.clone());
inherited
};
Ok(inherited.unwrap_or(isolated))
}
/// The checkpoint commit for one attempt's result. `Ok(Some)` is the
/// note to record, `Ok(None)` nothing to record, `Err` the fatal
/// failure message.
async fn snapshot(
&self,
context: &HookContext,
scope: ScopeId,
key: CheckpointKey,
node: &str,
status: &Status,
origin: ResultOrigin,
) -> Result<Option<Note>, String> {
if !self.host_workspaces {
if !self.unreachable_noted.swap(true, Ordering::SeqCst) {
warn!(
run_id = %self.run_id,
"the run's workspaces are not on this host; no checkpoint snapshot is taken"
);
}
return Ok(Some(Note::new(
CHECKPOINT_NOTE,
json!({
"execution": key.execution,
"firing": key.firing,
"attempt": key.attempt,
"skipped": "the workspace is not on this host",
}),
)));
}
let workspace = self.workspace_of(context, scope).await?;
if !self.workspaces.workspace_exists(&workspace).await {
// A skipped node or a driver-made outcome may precede the scope's
// environment; nothing of the stage's is on disk to snapshot.
if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) {
return Ok(Some(Note::new(
CHECKPOINT_NOTE,
json!({
"execution": key.execution,
"firing": key.firing,
"attempt": key.attempt,
"workspace": workspace,
"skipped": "the workspace does not exist yet",
}),
)));
}
return Err(format!(
"the workspace `{workspace}` of scope {scope} does not exist at {}",
self.workspaces.workspace_path(&workspace).display()
));
}
self.gate("commit", node).await;
let serialized = self.workspace_lock(&workspace);
let _held = serialized.lock().await;
match self
.workspaces
.commit(&workspace, key, node, status.tag())
.await
{
Ok(snapshot) => {
debug!(
run_id = %self.run_id,
node,
execution = key.execution,
firing = key.firing,
attempt = key.attempt,
reused = snapshot.reused,
"checkpoint committed"
);
lock(&self.committed).insert(key, (workspace.clone(), snapshot.sha.clone()));
Ok(Some(Note::new(
CHECKPOINT_NOTE,
json!({
"execution": key.execution,
"firing": key.firing,
"attempt": key.attempt,
"workspace": workspace,
"git_commit_sha": snapshot.sha,
"reused": snapshot.reused,
}),
)))
}
Err(error) => Err(format!(
"the checkpoint commit of `{node}` failed: {}",
collect_chain(&error).join(": ")
)),
}
}
/// The checkpoint's platform record, once per operation identity.
async fn record(
&self,
context: &HookContext,
scope: ScopeId,
key: CheckpointKey,
) -> Result<(), String> {
self.recorded_loaded
.get_or_try_init(|| self.load_recorded())
.await?;
if lock(&self.recorded).contains(&key) {
return Ok(());
}
let committed = lock(&self.committed).get(&key).cloned();
let (workspace, sha) = if let Some(committed) = committed {
committed
} else {
let workspace = self.workspace_of(context, scope).await?;
let serialized = self.workspace_lock(&workspace);
let held = serialized.lock().await;
let found = self.workspaces.find(&workspace, key).await;
drop(held);
let sha = found
.map_err(|error| {
format!(
"the checkpoint commit could not be looked up: {}",
collect_chain(&error).join(": ")
)
})?
.ok_or_else(|| {
format!(
"no checkpoint commit exists for execution {} firing {} attempt {}",
key.execution, key.firing, key.attempt
)
})?;
(workspace, sha)
};
let record = PlatformRecord::Checkpoint(CheckpointRecord {
execution: key.execution,
firing: key.firing,
attempt: Some(key.attempt),
workspace: Some(workspace),
git_commit_sha: Some(sha),
diff_summary: None,
patch_blob: None,
operation: Some(key.operation()),
});
self.records
.append(
&self.run_id,
&record,
Some(StagePosition {
execution: key.execution,
firing: key.firing,
}),
)
.await
.map_err(|error| {
format!(
"the checkpoint record could not be written: {}",
collect_chain(&error).join(": ")
)
})?;
lock(&self.recorded).insert(key);
Ok(())
}
/// The checkpoints already recorded for the run, read once: what a
/// resume's reissued routing decisions must not record again.
async fn load_recorded(&self) -> Result<(), String> {
let stored = self
.records
.read_kind(&self.run_id, PlatformRecordKind::Checkpoint)
.await
.map_err(|error| {
format!(
"the run's checkpoint records could not be read: {}",
collect_chain(&error).join(": ")
)
})?;
let mut recorded = lock(&self.recorded);
for record in stored {
let PlatformRecord::Checkpoint(checkpoint) = &record.record else {
continue;
};
if let Some(key) = checkpoint
.operation
.as_ref()
.and_then(CheckpointKey::from_operation)
{
recorded.insert(key);
}
}
Ok(())
}
/// Hold at a test gate when one is set for this point and node.
async fn gate(&self, point: &str, node: &str) {
let Some(dir) = &self.test_gates else {
return;
};
let hold = dir.join(format!("{point}.{node}.hold"));
if !fs::try_exists(&hold).await.unwrap_or(false) {
return;
}
let release = dir.join(format!("{point}.{node}.release"));
info!(point, node, "checkpoint held at a test gate");
while !fs::try_exists(&release).await.unwrap_or(false) {
time::sleep(GATE_POLL).await;
}
info!(point, node, "checkpoint released by its test gate");
}
}
fn lock<T>(mutex: &Mutex<T>) -> MutexGuard<'_, T> {
mutex.lock().unwrap_or_else(PoisonError::into_inner)
}
fn is_checkpoint_failure(status: &Status) -> bool {
matches!(status, Status::Failure(info) if info.class.as_str() == CHECKPOINT_FAILED_CLASS)
}
#[async_trait::async_trait]
impl ExecutionHooks for FabroHooks {
async fn before_attempt(
&self,
context: &HookContext,
request: AdmitAttempt,
) -> AttemptDecision {
self.inner.before_attempt(context, request).await
}
async fn prepare_result(
&self,
context: &HookContext,
request: PrepareResult,
) -> Result<Prepared, PrepareError> {
let node = request.view.node_name().to_owned();
let scope = request.view.scope;
let key = CheckpointKey {
execution: context.execution.raw(),
firing: request.view.firing.raw(),
attempt: request.view.attempt.raw(),
};
let original = request.outcome.status.clone();
let origin = request.origin;
let mut prepared = self.inner.prepare_result(context, request).await?;
let effective = prepared.adjustment.status.clone().unwrap_or(original);
if matches!(effective, Status::Cancelled) {
return Ok(prepared);
}
match self
.snapshot(context, scope, key, &node, &effective, origin)
.await
{
Ok(Some(note)) => prepared.notes.push(note),
Ok(None) => {}
Err(message) => {
warn!(
run_id = %self.run_id,
node,
execution = key.execution,
firing = key.firing,
attempt = key.attempt,
error = %message,
"checkpoint failed; the run ends"
);
self.fail_run(&message);
prepared.adjustment.status = Some(Status::Failure(
FailureInfo::new(message.clone()).with_class(CHECKPOINT_FAILED_CLASS),
));
prepared.adjustment.reason = Some(message);
}
}
Ok(prepared)
}
async fn after_record(&self, context: &HookContext, recorded: Recorded) -> Vec<Note> {
self.inner.after_record(context, recorded).await
}
async fn transition(
&self,
context: &HookContext,
transition: Transition,
) -> Result<TransitionReport, TransitionError> {
if is_checkpoint_failure(&transition.outcome.status) {
return Err(TransitionError::new(
"the stage's checkpoint commit failed; no route is taken",
));
}
let node = transition.view.node_name().to_owned();
let scope = transition.view.scope;
let key = CheckpointKey {
execution: context.execution.raw(),
firing: transition.view.firing.raw(),
attempt: transition.view.attempt.raw(),
};
let mut problems = Vec::new();
if self.host_workspaces {
self.gate("record", &node).await;
if let Err(problem) = self.record(context, scope, key).await {
warn!(
run_id = %self.run_id,
node,
execution = key.execution,
firing = key.firing,
error = %problem,
"the checkpoint record was not written"
);
problems.push(problem);
}
}
let mut report = self.inner.transition(context, transition).await?;
report.problems.extend(problems);
Ok(report)
}
async fn run_finished(&self, context: &HookContext, finished: RunFinished) -> Vec<Note> {
info!(
run_id = %self.run_id,
status = ?finished.status,
failure = finished.failure.as_deref().unwrap_or(""),
"Petri run finished; running the run-end hooks"
);
self.inner.run_finished(context, finished).await
}
async fn scope_released(&self, context: &HookContext, released: ScopeReleased) -> Vec<Note> {
debug!(
run_id = %self.run_id,
scope = %released.scope,
outcome = ?released.outcome,
"scope released; running the sandbox cleanup hooks"
);
self.inner.scope_released(context, released).await
}
}

View file

@ -33,7 +33,17 @@
//! - [`projection`] and [`projector`]: the view of a Petri run, folded from its
//! records and Fabro's platform records, and the pass that writes it after
//! each committed record;
//! - the platform adapters still to come: hooks and the run tools.
//! - [`hooks`]: Fabro's `ExecutionHooks`, the checkpoint commit in
//! `prepare_result` and its platform record in `transition`, around Petri's
//! own hook service for `[[run.hooks]]`;
//! - [`checkpoint`]: the Git snapshots of a run's host workspaces and the
//! snapshot repository they are published to;
//! - [`recovery`]: the resume-on-restart protocol, which brings every live
//! workspace to the snapshot its durable state names before the run goes back
//! to a worker;
//! - [`platform_records`]: Fabro's platform records as the adapters reach them,
//! in the server's database or over its API from a worker;
//! - the platform adapter still to come: the run tools.
//!
//! The Petri packages are pinned by revision in the workspace `Cargo.toml`
//! under `petri_*` keys.
@ -41,17 +51,22 @@
pub mod admission;
pub mod blobs;
pub mod check;
pub mod checkpoint;
pub mod engine;
pub mod hooks;
pub mod http_store;
pub mod interview;
pub mod petri;
pub mod platform_records;
pub mod projection;
pub mod projector;
pub mod recovery;
pub mod run_store;
pub mod runtime;
pub mod secrets;
#[cfg(feature = "test-support")]
pub mod test_support;
pub mod workspace;
pub use http_store::HttpRunStore;
pub use run_store::SqliteRunStore;

View file

@ -0,0 +1,188 @@
//! Fabro's platform records as a Petri run's adapters reach them.
//!
//! The records themselves are `fabro_store::platform_records`: one table
//! beside Petri's records, one typed enum of kinds. What differs is where
//! the adapter runs. In the server process (the in-process test path, and
//! startup recovery) the table is reached directly, through
//! [`SqlitePlatformRecords`]; in a run's worker process it is reached over
//! the server's API with the worker's token, through
//! [`HttpPlatformRecords`], as the run's Petri records are. Both answer
//! the one [`PlatformRecords`] interface the hooks and recovery use.
use std::fmt;
use std::sync::Arc;
use async_trait::async_trait;
use fabro_api::types::{PetriPlatformRecord, PetriPlatformRecordAppendRequest};
use fabro_client::Client;
use fabro_store::{
PlatformRecord, PlatformRecordKind, PlatformRecordStore, RunSummaryStore, StagePosition,
StoredPlatformRecord,
};
use fabro_types::RunId;
use serde_json::Value;
/// Why a platform record could not be stored or read.
#[derive(Debug, thiserror::Error)]
pub enum PlatformRecordError {
#[error("the platform record store failed")]
Store(#[source] fabro_store::Error),
#[error("the platform record request to the server failed")]
Api(#[source] anyhow::Error),
#[error("the platform record does not encode as JSON")]
Encode(#[source] serde_json::Error),
}
/// The platform records of a run, wherever the adapter runs.
#[async_trait]
pub trait PlatformRecords: Send + Sync {
/// Store a record at the run's next seq, tied to a Petri stage when it
/// belongs to one.
async fn append(
&self,
run_id: &RunId,
record: &PlatformRecord,
position: Option<StagePosition>,
) -> Result<StoredPlatformRecord, PlatformRecordError>;
/// The run's records of one kind, in seq order.
async fn read_kind(
&self,
run_id: &RunId,
kind: PlatformRecordKind,
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError>;
}
/// The table in the server's database, with the projector's wake-up after
/// each append.
#[derive(Clone)]
pub struct SqlitePlatformRecords {
store: PlatformRecordStore,
summaries: Arc<RunSummaryStore>,
}
impl fmt::Debug for SqlitePlatformRecords {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("SqlitePlatformRecords")
.finish_non_exhaustive()
}
}
impl SqlitePlatformRecords {
#[must_use]
pub fn new(summaries: Arc<RunSummaryStore>) -> Self {
Self {
store: summaries.platform_records(),
summaries,
}
}
}
#[async_trait]
impl PlatformRecords for SqlitePlatformRecords {
async fn append(
&self,
run_id: &RunId,
record: &PlatformRecord,
position: Option<StagePosition>,
) -> Result<StoredPlatformRecord, PlatformRecordError> {
let stored = self
.store
.append(run_id, record, position)
.await
.map_err(PlatformRecordError::Store)?;
self.summaries.notify_platform_record(*run_id);
Ok(stored)
}
async fn read_kind(
&self,
run_id: &RunId,
kind: PlatformRecordKind,
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
self.store
.read_kind(run_id, kind)
.await
.map_err(PlatformRecordError::Store)
}
}
/// The table as a run's worker reaches it: the server's
/// `/api/v1/runs/{id}/petri/platform-records` endpoints with the worker's
/// token.
pub struct HttpPlatformRecords {
client: Client,
}
impl fmt::Debug for HttpPlatformRecords {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("HttpPlatformRecords")
.field("server", &self.client.base_url())
.finish_non_exhaustive()
}
}
impl HttpPlatformRecords {
#[must_use]
pub fn new(client: Client) -> Self {
Self { client }
}
}
#[async_trait]
impl PlatformRecords for HttpPlatformRecords {
async fn append(
&self,
run_id: &RunId,
record: &PlatformRecord,
position: Option<StagePosition>,
) -> Result<StoredPlatformRecord, PlatformRecordError> {
let Value::Object(record) =
serde_json::to_value(record).map_err(PlatformRecordError::Encode)?
else {
return Err(PlatformRecordError::Api(anyhow::anyhow!(
"a platform record encodes as a JSON object"
)));
};
let body = PetriPlatformRecordAppendRequest {
record,
execution: position.map(|position| position.execution),
firing: position.map(|position| position.firing),
};
let stored = self
.client
.append_petri_platform_record(run_id, body)
.await
.map_err(PlatformRecordError::Api)?;
decode(stored)
}
async fn read_kind(
&self,
run_id: &RunId,
kind: PlatformRecordKind,
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
let records = self
.client
.list_petri_platform_records(run_id, Some(&kind.to_string()))
.await
.map_err(PlatformRecordError::Api)?;
records.into_iter().map(decode).collect()
}
}
/// A wire record back into the store's shape.
fn decode(wire: PetriPlatformRecord) -> Result<StoredPlatformRecord, PlatformRecordError> {
let record: PlatformRecord =
serde_json::from_value(Value::Object(wire.record)).map_err(PlatformRecordError::Encode)?;
let position = match (wire.execution, wire.firing) {
(Some(execution), Some(firing)) => Some(StagePosition { execution, firing }),
_ => None,
};
Ok(StoredPlatformRecord {
seq: wire.seq,
recorded_at: wire.recorded_at,
record,
position,
})
}

View file

@ -0,0 +1,450 @@
//! Resume on restart: the recovery protocol whose rule is that the
//! workspace a resumed stage sees matches Petri's durable execution state
//! (the integration plan's F3.5).
//!
//! For a Petri run the server finds in flight at startup, once the previous
//! worker's lease is released, [`recover`] reads the durable execution
//! state through `inspect_run` and decides:
//!
//! - a run with a `checkpoint_failed` finish anywhere is reported failed and
//! not resumed: a failed checkpoint cancelled it, and nothing of it is
//! reconciled;
//! - otherwise, for every live execution, the last durable finish names the
//! snapshot its workspace must sit on: the checkpoint record's commit, or,
//! when the record was lost to the crash, the commit found by its key in the
//! workspace's snapshot repository or history, which is then recorded again;
//! - a workspace that survives is verified to sit on that commit, unchanged, or
//! reset to it; a workspace that is gone is restored from the run's snapshot
//! repository into a fresh directory;
//! - a durable finish with no snapshot fails the run with a named error rather
//! than resume it on stale files.
//!
//! Every child invocation's scope has its own snapshots, keyed by
//! execution; a nested invocation that inherits its caller's sandbox shares
//! the caller's workspace, and the workspace is brought to the newest of
//! the live executions' snapshots on it.
//!
//! A run whose workspaces are not on this host (Docker, Daytona) is resumed
//! on its retained sandbox as it was left: the snapshot side of the
//! protocol reaches only host workspaces.
use std::collections::BTreeMap;
use std::path::PathBuf;
use std::sync::Arc;
use fabro_checkpoint::author::GitAuthor;
use fabro_store::platform_records::CheckpointRecord;
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace};
use fabro_types::{RunId, SandboxProviderKind};
use petri_execution::host::{self, HostError};
use petri_execution::inspect::{self, ExecutionInspection, InspectError};
use petri_execution::{Access, InvocationId, RunKey, RunStore};
use petri_store::StoreError;
use tracing::{info, warn};
use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces};
use crate::platform_records::{PlatformRecordError, PlatformRecords};
use crate::workspace::{WorkspaceLookup, WorkspaceLookupError};
/// What recovery needs: the run, where its workspaces are, its records.
pub struct RecoveryRequest {
pub run_id: RunId,
/// The run directory Petri ran under (the run's `petri` scratch).
pub run_dir: PathBuf,
pub store: Arc<dyn RunStore>,
pub records: Arc<dyn PlatformRecords>,
pub author: GitAuthor,
pub checkpoint: RunCheckpointSettings,
/// Whether the run's workspaces are on this host.
pub host_workspaces: bool,
}
impl RecoveryRequest {
/// The request a run's settings give: its Git author, its checkpoint
/// settings, and whether its sandbox provider keeps workspaces on this
/// host.
#[must_use]
pub fn for_run(
run_id: RunId,
run_dir: PathBuf,
store: Arc<dyn RunStore>,
records: Arc<dyn PlatformRecords>,
settings: &RunNamespace,
) -> Self {
Self {
run_id,
run_dir,
store,
records,
author: settings
.git
.author
.as_ref()
.map(GitAuthor::from)
.unwrap_or_default(),
checkpoint: settings.checkpoint.clone(),
host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL,
}
}
}
/// What was done to one workspace.
#[derive(Clone, Debug, PartialEq, Eq)]
pub enum WorkspaceAction {
/// It sat on the snapshot, unchanged.
Verified,
/// It was brought back to the snapshot.
Reset,
/// It was gone and was recreated from the snapshot repository.
Restored,
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct RecoveredWorkspace {
pub workspace: String,
pub sha: String,
pub action: WorkspaceAction,
}
/// What the server does with the run next.
#[derive(Clone, Debug, PartialEq, Eq)]
pub enum Recovery {
/// The store never held the run: it starts from its admitted graphs.
Start,
/// The run continues from its records, its live workspaces on their
/// snapshots.
Resume { workspaces: Vec<RecoveredWorkspace> },
/// The run cannot continue and is reported failed.
Failed { reason: String },
}
/// Why recovery could not decide: the records could not be read, or a
/// workspace could not be brought to its snapshot.
#[derive(Debug, thiserror::Error)]
pub enum RecoveryError {
#[error("the run's record could not be opened")]
Open(#[source] StoreError),
#[error("the run's coordinator log could not be read")]
Log(#[source] petri_execution::StoreError),
#[error("the run's coordinator state could not be read")]
State(#[source] HostError),
#[error("the run's record could not be inspected")]
Inspect(#[source] InspectError),
#[error("the run's workspaces could not be named")]
Lookup(#[source] WorkspaceLookupError),
#[error("the run's checkpoint records could not be read or written")]
Records(#[source] PlatformRecordError),
#[error("the workspace `{workspace}` could not be brought to its snapshot")]
Workspace {
workspace: String,
#[source]
source: CheckpointError,
},
}
/// One live execution's last durable finish and the snapshot it names.
struct Target {
execution: u64,
key: CheckpointKey,
}
/// Decide how the run continues, and bring its workspaces to their
/// snapshots.
pub async fn recover(request: RecoveryRequest) -> Result<Recovery, RecoveryError> {
let key = RunKey::new(request.run_id.to_string());
let logs = match request.store.open(&key, Access::Read).await {
Ok(logs) => logs,
Err(StoreError::NotFound { .. }) => return Ok(Recovery::Start),
Err(error) => return Err(RecoveryError::Open(error)),
};
// A record with no root invocation (the worker died between creating
// the run and declaring it) has nothing to reconcile; the worker's
// resume reports it as such.
let records = petri_execution::read_coordinator_log(&*logs)
.await
.map_err(RecoveryError::Log)?;
if records.is_empty() {
return Ok(Recovery::Resume {
workspaces: Vec::new(),
});
}
let state = host::stored_state(&*logs)
.await
.map_err(RecoveryError::State)?;
if !state.invocations.contains_key(&InvocationId::ROOT) {
return Ok(Recovery::Resume {
workspaces: Vec::new(),
});
}
let inspection = inspect::inspect_run(&*logs)
.await
.map_err(RecoveryError::Inspect)?;
drop(logs);
if let Some(failed) = checkpoint_failure(&inspection.executions) {
return Ok(Recovery::Failed { reason: failed });
}
if !request.host_workspaces {
warn!(
run_id = %request.run_id,
"the run's workspaces are not on this host; resuming on the retained sandbox as it was left"
);
return Ok(Recovery::Resume {
workspaces: Vec::new(),
});
}
let workspaces = RunWorkspaces::new(
request.run_dir.clone(),
request.run_id.to_string(),
request.author.clone(),
&request.checkpoint,
);
let lookup = WorkspaceLookup::new(Arc::clone(&request.store), key);
let recorded = recorded_checkpoints(&*request.records, &request.run_id).await?;
// The snapshot each live execution's workspace must sit on. A live
// execution is one whose log records no exit: `inspect_run` reports it
// as incomplete.
let mut candidates: BTreeMap<String, Vec<(Target, String)>> = BTreeMap::new();
for execution in inspection
.executions
.iter()
.filter(|execution| execution.status == "incomplete")
{
let Some(target) = last_finish(execution) else {
continue;
};
let owned = lookup
.of_invocation(execution.invocation)
.await
.map_err(RecoveryError::Lookup)?;
if owned.is_empty() {
continue;
}
let mut found = false;
for workspace in owned {
let sha = match recorded.get(&target.key) {
Some((recorded_workspace, sha))
if recorded_workspace.as_deref().is_none_or(|w| w == workspace) =>
{
Some(sha.clone())
}
_ => {
let sha = workspaces
.find(&workspace, target.key)
.await
.map_err(|source| RecoveryError::Workspace {
workspace: workspace.clone(),
source,
})?;
if let Some(sha) = &sha {
reconcile_record(
&*request.records,
&request.run_id,
target.key,
&workspace,
sha,
)
.await?;
}
sha
}
};
if let Some(sha) = sha {
found = true;
candidates
.entry(workspace)
.or_default()
.push((Target { ..target }, sha));
}
}
if !found {
return Ok(Recovery::Failed {
reason: format!(
"no checkpoint snapshot exists for the last durable finish of execution {} \
(firing {} attempt {}); the run cannot resume on stale files",
target.execution, target.key.firing, target.key.attempt
),
});
}
}
let mut recovered = Vec::new();
for (workspace, targets) in candidates {
let sha = newest(&workspaces, &workspace, &targets).await?;
let action = bring_to(&workspaces, &workspace, &sha, &targets).await?;
info!(
run_id = %request.run_id,
workspace,
sha,
action = ?action,
"workspace brought to its durable snapshot"
);
recovered.push(RecoveredWorkspace {
workspace,
sha,
action,
});
}
Ok(Recovery::Resume {
workspaces: recovered,
})
}
/// The reason a run with a failed checkpoint is reported failed, when it
/// has one.
fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option<String> {
executions.iter().find_map(|execution| {
execution
.engine
.as_ref()?
.attempts
.iter()
.find_map(|attempt| {
let failure = attempt.failure.as_ref()?;
(failure.class.as_str() == CHECKPOINT_FAILED_CLASS).then(|| {
format!(
"the checkpoint of {} (execution {} firing {} attempt {}) failed: {}",
attempt.node.as_deref().unwrap_or("a stage"),
execution.execution,
attempt.firing,
attempt.attempt,
failure.message
)
})
})
})
}
/// The last `StepFinished` of an execution's log.
fn last_finish(execution: &ExecutionInspection) -> Option<Target> {
let attempt = execution.engine.as_ref()?.attempts.last()?;
Some(Target {
execution: execution.execution.raw(),
key: CheckpointKey {
execution: execution.execution.raw(),
firing: attempt.firing,
attempt: attempt.attempt,
},
})
}
/// The run's checkpoint records by key: the workspace they name and the
/// commit.
async fn recorded_checkpoints(
records: &dyn PlatformRecords,
run_id: &RunId,
) -> Result<BTreeMap<CheckpointKey, (Option<String>, String)>, RecoveryError> {
let stored = records
.read_kind(run_id, PlatformRecordKind::Checkpoint)
.await
.map_err(RecoveryError::Records)?;
let mut recorded = BTreeMap::new();
for record in stored {
let PlatformRecord::Checkpoint(checkpoint) = record.record else {
continue;
};
let key = checkpoint
.operation
.as_ref()
.and_then(CheckpointKey::from_operation);
if let (Some(key), Some(sha)) = (key, checkpoint.git_commit_sha) {
recorded.insert(key, (checkpoint.workspace, sha));
}
}
Ok(recorded)
}
/// Write the record a crash lost, from the commit found by its key.
async fn reconcile_record(
records: &dyn PlatformRecords,
run_id: &RunId,
key: CheckpointKey,
workspace: &str,
sha: &str,
) -> Result<(), RecoveryError> {
info!(
run_id = %run_id,
execution = key.execution,
firing = key.firing,
attempt = key.attempt,
sha,
"checkpoint record reconciled from the run branch"
);
let record = PlatformRecord::Checkpoint(CheckpointRecord {
execution: key.execution,
firing: key.firing,
attempt: Some(key.attempt),
workspace: Some(workspace.to_string()),
git_commit_sha: Some(sha.to_string()),
diff_summary: None,
patch_blob: None,
operation: Some(key.operation()),
});
records
.append(
run_id,
&record,
Some(StagePosition {
execution: key.execution,
firing: key.firing,
}),
)
.await
.map_err(RecoveryError::Records)?;
Ok(())
}
/// Of the snapshots live executions name on one workspace, the one every
/// other descends from, else the last named.
async fn newest(
workspaces: &RunWorkspaces,
workspace: &str,
targets: &[(Target, String)],
) -> Result<String, RecoveryError> {
let mut chosen = &targets[0].1;
for (_, sha) in &targets[1..] {
if workspaces
.is_ancestor(workspace, chosen, sha)
.await
.map_err(|source| RecoveryError::Workspace {
workspace: workspace.to_string(),
source,
})?
{
chosen = sha;
}
}
Ok(chosen.clone())
}
/// Verify, reset or restore the workspace onto `sha`.
async fn bring_to(
workspaces: &RunWorkspaces,
workspace: &str,
sha: &str,
targets: &[(Target, String)],
) -> Result<WorkspaceAction, RecoveryError> {
let failed = |source| RecoveryError::Workspace {
workspace: workspace.to_string(),
source,
};
if workspaces.workspace_exists(workspace).await {
if workspaces.matches(workspace, sha).await.map_err(failed)? {
return Ok(WorkspaceAction::Verified);
}
workspaces.reset(workspace, sha).await.map_err(failed)?;
return Ok(WorkspaceAction::Reset);
}
let key = targets
.iter()
.find(|(_, candidate)| candidate == sha)
.map_or(targets[0].0.key, |(target, _)| target.key);
workspaces
.restore(workspace, key, sha)
.await
.map_err(failed)?;
Ok(WorkspaceAction::Restored)
}

View file

@ -1,5 +1,71 @@
//! Petri's test kit, for Fabro crates that check a store implementation
//! against Petri's contract from their own tests. Compiled only with the
//! against Petri's contract from their own tests, and an in-memory platform
//! record store for tests of the hooks and recovery. Compiled only with the
//! `test-support` feature, which a dev-dependency turns on.
use std::collections::HashMap;
use std::sync::{Mutex, MutexGuard, PoisonError};
use async_trait::async_trait;
use fabro_store::platform_records::now_ms;
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord};
use fabro_types::RunId;
pub use petri_testkit::run_store;
use crate::platform_records::{PlatformRecordError, PlatformRecords};
/// Platform records kept in memory, per run, in seq order.
#[derive(Debug, Default)]
pub struct MemoryPlatformRecords {
runs: Mutex<HashMap<RunId, Vec<StoredPlatformRecord>>>,
}
impl MemoryPlatformRecords {
#[must_use]
pub fn new() -> Self {
Self::default()
}
/// Every record of the run, in seq order.
#[must_use]
pub fn records(&self, run_id: &RunId) -> Vec<StoredPlatformRecord> {
lock(&self.runs).get(run_id).cloned().unwrap_or_default()
}
}
fn lock<T>(mutex: &Mutex<T>) -> MutexGuard<'_, T> {
mutex.lock().unwrap_or_else(PoisonError::into_inner)
}
#[async_trait]
impl PlatformRecords for MemoryPlatformRecords {
async fn append(
&self,
run_id: &RunId,
record: &PlatformRecord,
position: Option<StagePosition>,
) -> Result<StoredPlatformRecord, PlatformRecordError> {
let mut runs = lock(&self.runs);
let records = runs.entry(*run_id).or_default();
let stored = StoredPlatformRecord {
seq: records.len() as u64 + 1,
recorded_at: now_ms(),
record: record.clone(),
position,
};
records.push(stored.clone());
Ok(stored)
}
async fn read_kind(
&self,
run_id: &RunId,
kind: PlatformRecordKind,
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
Ok(self
.records(run_id)
.into_iter()
.filter(|record| record.record.kind() == kind)
.collect())
}
}

View file

@ -0,0 +1,116 @@
//! Which workspace a Petri scope runs in, from the run's own records.
//!
//! Petri names an isolated scope's workspace after its invocation and scope
//! (`invocation-<n>-scope-<m>`), and a nested invocation that inherits its
//! caller's sandbox shares the caller's workspace through the lease the
//! coordinator recorded. The hooks and recovery both need the workspace id
//! behind a scope, and both read it the same way here: the direct name
//! when its workspace exists on this host, else the lease the invocation's
//! declaration names, resolved through the resource log.
use std::sync::Arc;
use petri_execution::host::{self, HostError};
use petri_execution::{
Access, InvocationId, ResourceError, ResourceStore, RunKey, RunLogs, RunStore, SandboxBinding,
};
use petri_runtime::executor::WorkspaceId;
use petri_runtime::ir::ScopeId;
use petri_store::StoreError;
/// Why a workspace could not be named from the run's records.
#[derive(Debug, thiserror::Error)]
pub enum WorkspaceLookupError {
#[error("the run's record could not be opened")]
Open(#[source] StoreError),
#[error("the run's coordinator state could not be read")]
State(#[source] HostError),
#[error("the run's resource log could not be read")]
Resources(#[source] ResourceError),
#[error("invocation {invocation} is not in the run's record")]
UnknownInvocation { invocation: InvocationId },
}
/// The workspace id of an isolated scope: what the coordinator allocates
/// for `scope` in `invocation`.
#[must_use]
pub fn isolated_workspace(invocation: InvocationId, scope: ScopeId) -> String {
WorkspaceId::scoped(Some(&invocation.workspace_prefix()), scope)
.as_str()
.to_owned()
}
/// The workspaces of a run, read from its records through a handle that
/// holds no lease.
pub struct WorkspaceLookup {
store: Arc<dyn RunStore>,
key: RunKey,
}
impl WorkspaceLookup {
#[must_use]
pub fn new(store: Arc<dyn RunStore>, key: RunKey) -> Self {
Self { store, key }
}
async fn logs(&self) -> Result<Arc<dyn RunLogs>, WorkspaceLookupError> {
self.store
.open(&self.key, Access::Read)
.await
.map_err(WorkspaceLookupError::Open)
}
/// The workspace an invocation inherited from its caller, or `None`
/// when the invocation owns its sandboxes.
pub async fn inherited(
&self,
invocation: InvocationId,
) -> Result<Option<String>, WorkspaceLookupError> {
let logs = self.logs().await?;
let state = host::stored_state(&*logs)
.await
.map_err(WorkspaceLookupError::State)?;
let declared = state
.invocations
.get(&invocation)
.ok_or(WorkspaceLookupError::UnknownInvocation { invocation })?;
match declared.declaration.sandbox {
SandboxBinding::Isolated => Ok(None),
SandboxBinding::Inherited { lease } => {
let resources = ResourceStore::load(&logs)
.await
.map_err(WorkspaceLookupError::Resources)?;
let record = resources
.resolve(lease)
.map_err(WorkspaceLookupError::Resources)?;
Ok(Some(record.workspace.as_str().to_owned()))
}
}
}
/// Every workspace an invocation runs in: its own leases' workspaces,
/// or the one it inherited.
pub async fn of_invocation(
&self,
invocation: InvocationId,
) -> Result<Vec<String>, WorkspaceLookupError> {
if let Some(inherited) = self.inherited(invocation).await? {
return Ok(vec![inherited]);
}
let logs = self.logs().await?;
let resources = ResourceStore::load(&logs)
.await
.map_err(WorkspaceLookupError::Resources)?;
let mut workspaces: Vec<String> = resources
.records()
.filter(|record| {
record.allocation.invocation == invocation
&& record.state != petri_execution::LeaseState::Deleted
})
.map(|record| record.workspace.as_str().to_owned())
.collect();
workspaces.sort();
workspaces.dedup();
Ok(workspaces)
}
}

View file

@ -0,0 +1,691 @@
//! Fabro's hooks on a Petri run, in process: command-only bundles run
//! through `engine::run` on the host sandbox with the memory store and
//! in-memory platform records, and the checkpoint commit, its record, the
//! failure route, the fatal checkpoint, and the run-end hooks are checked
//! against the workspace's Git history and the run's records.
//!
//! Every run acquires its scope through the sandbox-driver host plugin, so
//! the tests skip when that executable is not found, unless
//! `FABRO_REQUIRE_SANDBOX_PLUGINS` is set. The crash cases of the recovery
//! protocol need a worker to kill and live in the CLI's scenario suite.
#![expect(
clippy::disallowed_methods,
reason = "the tests locate the plugin executable through the process environment and read the workspace's history with git"
)]
#![expect(clippy::print_stderr, reason = "a skipped test says why on its stderr")]
use std::collections::BTreeMap;
use std::env;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use fabro_checkpoint::author::GitAuthor;
use fabro_petri::admission::AdmittedGraphs;
use fabro_petri::check::{self, Bundle, CheckRequest, Launch};
use fabro_petri::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces};
use fabro_petri::engine::{self, Execution, RunRequest, RunStatus};
use fabro_petri::hooks::HooksSpec;
use fabro_petri::platform_records::PlatformRecords;
use fabro_petri::recovery::{self, Recovery, RecoveryRequest};
use fabro_petri::runtime::RuntimeSpec;
use fabro_petri::test_support::MemoryPlatformRecords;
use fabro_store::{PlatformRecord, PlatformRecordKind};
use fabro_types::settings::run::RunCheckpointSettings;
use fabro_types::{RunId, SandboxProviderKind};
use petri_execution::inspect::{self, RunInspection};
use petri_store::{Access, MemoryRunStore, RunKey, RunStore as _};
use tokio::fs;
use tokio::process::Command;
use tokio_util::sync::CancellationToken;
mod support;
const HOST_PLUGIN: &str = "sandbox-driver-host";
const HOST_PLUGIN_OVERRIDE: &str = "PETRI_SANDBOX_HOST_PLUGIN";
const REQUIRE_ENV: &str = "FABRO_REQUIRE_SANDBOX_PLUGINS";
/// The host plugin as Petri's lookup finds it: the override variable, else
/// the executable on `PATH`. `None`, after saying so, when the test should
/// skip; a panic when the environment forbids a skip.
fn host_plugin() -> Option<PathBuf> {
let found = env::var_os(HOST_PLUGIN_OVERRIDE)
.map(PathBuf::from)
.or_else(|| {
env::split_paths(&env::var_os("PATH")?)
.map(|dir| dir.join(HOST_PLUGIN))
.find(|candidate| candidate.is_file())
});
if found.is_none() {
assert!(
env::var_os(REQUIRE_ENV).is_none(),
"{REQUIRE_ENV} is set, but {HOST_PLUGIN} is not on PATH and {HOST_PLUGIN_OVERRIDE} is unset"
);
eprintln!("skipping: {HOST_PLUGIN} is not on PATH and {HOST_PLUGIN_OVERRIDE} is unset");
}
found
}
/// A command-only bundle: the stage lines go between `start` and `exit`,
/// the edge lines after them.
fn workflow(stages: &str, edges: &str) -> String {
format!(
"digraph Hooks {{\n graph [goal=\"Check the hooks\", default_max_retries=0]\n start \
[shape=Mdiamond]\n exit [shape=Msquare]\n{stages}\n{edges}\n}}\n"
)
}
const SETTINGS: &str = "_version = 1\n\n[workflow]\ngraph = \"workflow.fabro\"\n";
/// The bundle admitted the way the create handler admits it.
fn admit(workflow: &str, settings: &str) -> AdmittedGraphs {
let request = CheckRequest {
bundle: Bundle {
files: BTreeMap::from([
("workflow.fabro".to_string(), workflow.to_string()),
("workflow.toml".to_string(), settings.to_string()),
]),
entrypoint: "workflow.fabro".to_string(),
project_toml: None,
},
inputs: BTreeMap::new(),
launch: Launch::default(),
runtime: RuntimeSpec::default(),
};
let admitted = check::check(&request).expect("the bundle is admitted");
AdmittedGraphs {
graph: admitted.graph,
children: admitted.children,
}
}
/// One run's pieces: the store, its platform records, where it ran.
struct Harness {
run_id: RunId,
run_dir: PathBuf,
store: Arc<MemoryRunStore>,
records: Arc<MemoryPlatformRecords>,
_root: tempfile::TempDir,
}
impl Harness {
fn new() -> Self {
let root = tempfile::tempdir().expect("a temp dir");
Self {
run_id: RunId::new(),
run_dir: root.path().join("run"),
store: Arc::new(MemoryRunStore::new()),
records: Arc::new(MemoryPlatformRecords::new()),
_root: root,
}
}
fn hooks(&self) -> HooksSpec {
HooksSpec {
records: Arc::clone(&self.records) as Arc<dyn PlatformRecords>,
author: GitAuthor::default(),
checkpoint: RunCheckpointSettings::default(),
host_workspaces: true,
test_gates: None,
}
}
/// Run the bundle to its end through the engine module, as the worker
/// does, and report what the record says.
async fn run(&self, workflow: &str, settings: &str) -> engine::RunOutcome {
let (interviewer, observers) = no_questions();
let request = RunRequest {
run_id: self.run_id.to_string(),
run_dir: self.run_dir.clone(),
execution: Execution::Start(admit(workflow, settings)),
store: Arc::clone(&self.store) as Arc<dyn petri_store::RunStore>,
runtime: RuntimeSpec::default(),
provider: SandboxProviderKind::LOCAL,
cancel: CancellationToken::new(),
interviewer,
observers,
secrets: None,
blobs: None,
hooks: Some(self.hooks()),
};
engine::run(request).await.expect("the run executes")
}
async fn inspection(&self) -> RunInspection {
let logs = self
.store
.open(&RunKey::new(self.run_id.to_string()), Access::Read)
.await
.expect("the run opens for reading");
inspect::inspect_run(&*logs)
.await
.expect("the stored run inspects")
}
fn workspaces(&self) -> RunWorkspaces {
RunWorkspaces::new(
self.run_dir.clone(),
self.run_id.to_string(),
GitAuthor::default(),
&RunCheckpointSettings::default(),
)
}
/// The one workspace the run's root scope used.
async fn workspace(&self) -> String {
let mut entries = fs::read_dir(self.run_dir.join("scopes"))
.await
.expect("the scopes directory exists");
let mut names = Vec::new();
while let Some(entry) = entries.next_entry().await.expect("an entry reads") {
names.push(entry.file_name().to_string_lossy().into_owned());
}
assert_eq!(names.len(), 1, "one workspace: {names:?}");
names.remove(0)
}
fn workspace_path(&self, workspace: &str) -> PathBuf {
self.workspaces().workspace_path(workspace)
}
/// The checkpoint records, in seq order, as `(key, sha)`.
fn checkpoints(&self) -> Vec<(CheckpointKey, String)> {
self.records
.records(&self.run_id)
.into_iter()
.filter_map(|stored| match stored.record {
PlatformRecord::Checkpoint(record) => Some((
CheckpointKey::from_operation(record.operation.as_ref()?)?,
record.git_commit_sha?,
)),
_ => None,
})
.collect()
}
async fn recover(&self) -> Recovery {
recovery::recover(RecoveryRequest {
run_id: self.run_id,
run_dir: self.run_dir.clone(),
store: Arc::clone(&self.store) as Arc<dyn petri_store::RunStore>,
records: Arc::clone(&self.records) as Arc<dyn PlatformRecords>,
author: GitAuthor::default(),
checkpoint: RunCheckpointSettings::default(),
host_workspaces: true,
})
.await
.expect("recovery decides")
}
}
/// An interviewer for a run that asks nothing, with its expiry observer.
fn no_questions() -> (
Arc<dyn petri_execution::Interviewer>,
Vec<Arc<dyn petri_execution::ExecutionObserver>>,
) {
let interviewer = support::no_questions(Arc::new(support::Silent));
let observers = vec![interviewer.observer()];
(Arc::new(interviewer), observers)
}
/// `git` in a workspace, its stdout.
async fn git(path: &Path, args: &[&str]) -> String {
let output = Command::new("git")
.args(args)
.current_dir(path)
.output()
.await
.expect("git runs");
assert!(
output.status.success(),
"git {args:?} failed: {}",
String::from_utf8_lossy(&output.stderr)
);
String::from_utf8_lossy(&output.stdout).trim().to_string()
}
/// The commits on the run branch, oldest first, as `(sha, subject, key)`.
async fn commits(path: &Path) -> Vec<(String, String, Option<CheckpointKey>)> {
let log = git(path, &["log", "--reverse", "--format=%H%x00%s%x00%B%x1e"]).await;
log.split('\u{1e}')
.filter(|entry| !entry.trim().is_empty())
.map(|entry| {
let mut parts = entry.trim_start().splitn(3, '\0');
let sha = parts.next().unwrap_or_default().to_string();
let subject = parts.next().unwrap_or_default().to_string();
let body = parts.next().unwrap_or_default();
(sha, subject, CheckpointKey::from_message(body))
})
.collect()
}
/// The stages that ran, by node name, with their final status.
fn stages(inspection: &RunInspection) -> Vec<(String, String)> {
inspection
.executions
.iter()
.filter_map(|execution| execution.engine.as_ref())
.flat_map(|engine| engine.history.iter())
.map(|record| (record.node.to_string(), record.status.to_string()))
.collect()
}
/// Every finished stage is committed on the run branch with the identity
/// trailers, its platform record names the commit, and the run-end hooks
/// reached Petri's local service through Fabro's wrapper.
#[tokio::test]
async fn every_finish_is_committed_and_recorded() {
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
let workflow = workflow(
" write [shape=parallelogram, script=\"echo one > out.txt\"]\n check \
[shape=parallelogram, script=\"test \\\"$(cat out.txt)\\\" = one\"]",
" start -> write -> check -> exit",
);
let settings = format!(
"{SETTINGS}\n[[run.hooks]]\nevent = \"run_complete\"\nscript = \"echo run_complete >> \
run-end.log\"\n\n[[run.hooks]]\nevent = \"sandbox_cleanup\"\nscript = \"echo \
sandbox_cleanup >> run-end.log\"\n"
);
let outcome = harness.run(&workflow, &settings).await;
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
assert!(outcome.complete, "{:?}", outcome.incomplete);
let workspace = harness.workspace().await;
let path = harness.workspace_path(&workspace);
assert_eq!(
git(&path, &["rev-parse", "--abbrev-ref", "HEAD"]).await,
format!("fabro/run/{}", harness.run_id)
);
let commits = commits(&path).await;
let subjects: Vec<&str> = commits
.iter()
.map(|(_, subject, _)| subject.as_str())
.collect();
let run_id = harness.run_id.to_string();
assert_eq!(subjects, vec![
format!("fabro({run_id}): start (success)"),
format!("fabro({run_id}): write (success)"),
format!("fabro({run_id}): check (success)"),
format!("fabro({run_id}): exit (success)"),
]);
assert!(
commits.iter().all(|(_, _, key)| key.is_some()),
"every commit carries its key: {commits:?}"
);
let checkpoints = harness.checkpoints();
assert_eq!(checkpoints.len(), 4, "{checkpoints:?}");
let by_sha: Vec<&String> = checkpoints.iter().map(|(_, sha)| sha).collect();
let committed: Vec<&String> = commits.iter().map(|(sha, _, _)| sha).collect();
assert_eq!(by_sha, committed, "each record names its stage's commit");
for ((key, _), (_, _, trailer)) in checkpoints.iter().zip(&commits) {
assert_eq!(Some(*key), *trailer);
}
assert!(
checkpoints.iter().all(|(key, _)| key.execution == 0),
"{checkpoints:?}"
);
// The snapshot repository holds every checkpoint.
let published = harness
.workspaces()
.published(&workspace)
.await
.expect("the snapshots list");
assert_eq!(published.len(), 4);
// `run_complete` and `sandbox_cleanup` ran through the forwarded
// service, with the sandbox in place.
let run_end = fs::read_to_string(path.join("run-end.log"))
.await
.expect("the run-end hooks wrote their log");
assert_eq!(run_end, "run_complete\nsandbox_cleanup\n");
}
/// A stage that fails on its own terms is committed like a successful one,
/// and its failure route runs on the committed files.
#[tokio::test]
async fn a_failed_stage_is_committed_and_its_route_sees_the_files() {
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
let workflow = workflow(
" work [shape=parallelogram, script=\"echo partial > out.txt; exit 1\"]\n fix \
[shape=parallelogram, script=\"test \\\"$(cat out.txt)\\\" = partial && echo fixed >> \
out.txt\"]",
" start -> work -> exit\n work -> fix [condition=\"outcome=failed\"]\n fix -> exit",
);
let outcome = harness.run(&workflow, SETTINGS).await;
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
let inspection = harness.inspection().await;
assert_eq!(stages(&inspection), vec![
("start".to_string(), "success".to_string()),
("work".to_string(), "failure".to_string()),
("fix".to_string(), "success".to_string()),
("exit".to_string(), "success".to_string()),
]);
let workspace = harness.workspace().await;
let path = harness.workspace_path(&workspace);
let commits = commits(&path).await;
let run_id = harness.run_id.to_string();
assert_eq!(commits[1].1, format!("fabro({run_id}): work (failure)"));
assert_eq!(
git(&path, &["show", &format!("{}:out.txt", commits[1].0)]).await,
"partial",
"the failed stage's files are in its snapshot"
);
assert_eq!(
git(&path, &["show", &format!("{}:out.txt", commits[2].0)]).await,
"partial\nfixed",
"the route ran on the committed files"
);
assert_eq!(harness.checkpoints().len(), 4);
}
/// A checkpoint commit that fails is fatal: the stage's outcome is recorded
/// as `checkpoint_failed`, no route is taken, the run ends failed with the
/// checkpoint's error, and a restart reports it failed without resuming.
#[tokio::test]
async fn a_failed_checkpoint_ends_the_run_with_no_route() {
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
let workflow = workflow(
" wreck [shape=parallelogram, script=\"rm -rf .git && echo garbage > .git && echo wrecked \
> out.txt\"]\n next [shape=parallelogram, script=\"echo next > next.txt\"]\n fix \
[shape=parallelogram, script=\"echo fix > fix.txt\"]",
" start -> wreck -> next -> exit\n wreck -> fix [condition=\"outcome=failed\"]\n fix -> \
exit",
);
let outcome = harness.run(&workflow, SETTINGS).await;
assert_eq!(outcome.status, RunStatus::Failed, "{outcome:?}");
let failure = outcome
.failure
.clone()
.expect("the run failed with a reason");
assert!(
failure.contains("checkpoint commit of `wreck` failed"),
"{failure}"
);
let inspection = harness.inspection().await;
let attempts: Vec<_> = inspection
.executions
.iter()
.filter_map(|execution| execution.engine.as_ref())
.flat_map(|engine| engine.attempts.iter())
.collect();
let wreck = attempts
.iter()
.find(|attempt| attempt.node.as_deref() == Some("wreck"))
.expect("the wrecked stage finished");
assert_eq!(wreck.status, "failure");
assert_eq!(
wreck.failure.as_ref().map(|failure| failure.class.as_str()),
Some(CHECKPOINT_FAILED_CLASS),
"{wreck:?}"
);
assert!(
attempts
.iter()
.all(|attempt| !matches!(attempt.node.as_deref(), Some("next" | "fix"))),
"no route ran: {:?}",
stages(&inspection)
);
let workspace = harness.workspace().await;
let path = harness.workspace_path(&workspace);
assert!(!fs::try_exists(path.join("next.txt")).await.expect("exists"));
assert!(!fs::try_exists(path.join("fix.txt")).await.expect("exists"));
// The wrecked stage has no record: its commit never landed.
let checkpoints = harness.checkpoints();
assert_eq!(checkpoints.len(), 1, "{checkpoints:?}");
// A restart finds the failed checkpoint and reports the run failed.
let recovery = harness.recover().await;
assert!(
matches!(&recovery, Recovery::Failed { reason } if reason.contains("checkpoint of wreck")),
"{recovery:?}"
);
}
/// The two ends of recovery that need no crash: a run the store never held
/// starts over, and a run that finished has nothing to bring back.
#[tokio::test]
async fn recovery_starts_an_unknown_run_and_resumes_a_finished_one() {
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
assert_eq!(harness.recover().await, Recovery::Start);
let workflow = workflow(
" write [shape=parallelogram, script=\"echo one > out.txt\"]",
" start -> write -> exit",
);
let outcome = harness.run(&workflow, SETTINGS).await;
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
assert_eq!(harness.recover().await, Recovery::Resume {
workspaces: Vec::new(),
});
assert_eq!(
harness
.records
.read_kind(&harness.run_id, PlatformRecordKind::Checkpoint)
.await
.expect("the records read")
.len(),
3
);
}
/// A `[[run.hooks]]` hook that blocks a tool effect keeps working through
/// the forwarded local service: the agent's `rm` is refused by the
/// `pre_tool_use` hook, the model is told why, and the file it aimed at is
/// still in the stage's snapshot.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn a_run_hook_blocks_a_tool_effect_through_the_forwarded_service() {
use fabro_auth::test_support::env_credential_source;
use fabro_llm::test_support::test_catalog_with_provider_base_url;
use fabro_petri::runtime;
use fabro_test::{TwinScenario, TwinScenarios, TwinToolCall};
use lithos_llm::catalog::ProviderId;
use serde_json::json;
const MODEL: &str = "gpt-5.6-sol";
if host_plugin().is_none() {
return;
}
let twin = fabro_test::twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
TwinScenarios::new(namespace.clone())
.scenario(
TwinScenario::responses(MODEL)
.input_contains("Remove the scratch file")
.tool_call(TwinToolCall::new(
"shell_command",
json!({ "command": "rm -f scratch.txt && echo removed" }),
)),
)
.scenario(
TwinScenario::responses(MODEL)
.input_contains("destructive commands are not allowed")
.text("Understood, the file stays."),
)
.load(twin)
.await;
let api_key = namespace.clone();
let credentials =
env_credential_source(move |name| (name == "OPENAI_API_KEY").then(|| api_key.clone()));
let client = runtime::model_client(
test_catalog_with_provider_base_url("openai", &twin.base_url),
credentials,
None,
&[ProviderId::new("openai")],
)
.expect("the model client builds")
.expect("openai is eligible");
let harness = Harness::new();
let (interviewer, observers) = no_questions();
let workflow = format!(
"digraph Hooks {{\n graph [backend=\"api\", goal=\"Check the tool hooks\", \
default_max_retries=0]\n start [shape=Mdiamond]\n exit [shape=Msquare]\n seed \
[shape=parallelogram, script=\"echo keep > scratch.txt\"]\n agent [prompt=\"Remove the \
scratch file with the shell tool.\", model=\"{MODEL}\", provider=\"openai\", \
fidelity=\"full\"]\n check [shape=parallelogram, script=\"test \\\"$(cat scratch.txt)\\\" \
= keep\"]\n start -> seed -> agent -> check -> exit\n}}\n"
);
let settings = format!(
"{SETTINGS}\n[[run.hooks]]\nname = \"no-destruction\"\nevent = \"pre_tool_use\"\nmatcher \
= \"shell\"\nscript = \"if grep -q 'rm ' \\\"$FABRO_HOOK_CONTEXT\\\"; then echo \
'{{\\\"decision\\\":\\\"block\\\",\\\"reason\\\":\\\"destructive commands are not \
allowed\\\"}}'; exit 2; fi\"\n"
);
let request = RunRequest {
run_id: harness.run_id.to_string(),
run_dir: harness.run_dir.clone(),
execution: Execution::Start(admit(&workflow, &settings)),
store: Arc::clone(&harness.store) as Arc<dyn petri_store::RunStore>,
runtime: RuntimeSpec {
model_client: Some(client),
..RuntimeSpec::default()
},
provider: SandboxProviderKind::LOCAL,
cancel: CancellationToken::new(),
interviewer,
observers,
secrets: None,
blobs: None,
hooks: Some(harness.hooks()),
};
let outcome = engine::run(request).await.expect("the run executes");
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
let inspection = harness.inspection().await;
assert_eq!(stages(&inspection), vec![
("start".to_string(), "success".to_string()),
("seed".to_string(), "success".to_string()),
("agent".to_string(), "success".to_string()),
("check".to_string(), "success".to_string()),
("exit".to_string(), "success".to_string()),
]);
let requests = twin.request_logs(&namespace).await;
let inputs: Vec<&str> = requests["requests"]
.as_array()
.map(|requests| {
requests
.iter()
.filter_map(|request| request["input_text"].as_str())
.collect()
})
.unwrap_or_default();
assert_eq!(inputs.len(), 2, "{inputs:?}");
assert!(
inputs[1].contains("destructive commands are not allowed"),
"the model was told why the tool was blocked: {inputs:?}"
);
let workspace = harness.workspace().await;
let path = harness.workspace_path(&workspace);
assert_eq!(
git(&path, &["show", "HEAD:scratch.txt"]).await,
"keep",
"the blocked removal never happened"
);
assert_eq!(harness.checkpoints().len(), 5);
}
/// The run's records name the workspace its root invocation ran in: what
/// recovery reads to find the workspace of a live execution.
#[tokio::test]
async fn the_records_name_the_root_invocations_workspace() {
use fabro_petri::workspace::WorkspaceLookup;
use petri_execution::InvocationId;
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
let workflow = workflow(
" write [shape=parallelogram, script=\"echo one > out.txt\"]",
" start -> write -> exit",
);
let outcome = harness.run(&workflow, SETTINGS).await;
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
let lookup = WorkspaceLookup::new(
Arc::clone(&harness.store) as Arc<dyn petri_store::RunStore>,
RunKey::new(harness.run_id.to_string()),
);
let named = lookup
.of_invocation(InvocationId::ROOT)
.await
.expect("the lookup reads the records");
assert_eq!(named, vec![harness.workspace().await]);
let inspection = harness.inspection().await;
let statuses: Vec<&str> = inspection
.executions
.iter()
.map(|execution| execution.status)
.collect();
assert_eq!(statuses, vec!["finished"]);
}
/// The branches of a parallel node share their caller's workspace: their
/// checkpoints are committed one at a time on the one run branch, every
/// firing of every execution gets its record, and no transition reports a
/// problem.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn parallel_branches_checkpoint_the_shared_workspace_in_turn() {
if host_plugin().is_none() {
return;
}
let harness = Harness::new();
let workflow = workflow(
" fork [shape=component]\n a [shape=parallelogram, script=\"echo a > a.txt\"]\n b \
[shape=parallelogram, script=\"echo b > b.txt\"]\n merge [shape=tripleoctagon]\n check \
[shape=parallelogram, script=\"test -f a.txt && test -f b.txt\"]",
" start -> fork\n fork -> a\n fork -> b\n a -> merge\n b -> merge\n merge -> check -> \
exit",
);
let outcome = harness.run(&workflow, SETTINGS).await;
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
let records = support::all_records(harness.store.as_ref(), &harness.run_id.to_string()).await;
let problems: Vec<&serde_json::Value> = records
.iter()
.filter_map(|record| record.pointer("/event/$note"))
.filter(|note| note["kind"] == "transition" && !note["payload"]["problems"].is_null())
.collect();
assert!(problems.is_empty(), "transition problems: {problems:#?}");
let inspection = harness.inspection().await;
let mut finished: Vec<CheckpointKey> = inspection
.executions
.iter()
.filter_map(|execution| Some((execution.execution.raw(), execution.engine.as_ref()?)))
.flat_map(|(execution, engine)| {
engine.attempts.iter().map(move |attempt| CheckpointKey {
execution,
firing: attempt.firing,
attempt: attempt.attempt,
})
})
.collect();
finished.sort();
let mut recorded: Vec<CheckpointKey> = harness
.checkpoints()
.into_iter()
.map(|(key, _)| key)
.collect();
recorded.sort();
assert_eq!(
recorded, finished,
"every finish of every execution is recorded"
);
assert_eq!(inspection.executions.len(), 3, "the root and two branches");
assert_eq!(harness.workspace().await, "invocation-0-scope-0");
}

View file

@ -117,6 +117,7 @@ pub(crate) fn run_request(
interviewer: Arc::new(interviewer),
secrets: None,
blobs: None,
hooks: None,
}
}

View file

@ -376,6 +376,13 @@ pub struct GitIdentityRecord {
pub struct CheckpointRecord {
pub execution: u64,
pub firing: u64,
/// The attempt whose files the commit holds; absent on a record written
/// before the field existed.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub attempt: Option<u32>,
/// The Petri workspace id the commit was made in.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub workspace: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub git_commit_sha: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
@ -857,6 +864,8 @@ mod tests {
PlatformRecordKind::Checkpoint => PlatformRecord::Checkpoint(CheckpointRecord {
execution: 0,
firing: 3,
attempt: Some(1),
workspace: Some("invocation-0-scope-0".to_string()),
git_commit_sha: Some("def".to_string()),
diff_summary: Some(DiffSummary {
files_changed: 1,

View file

@ -261,7 +261,8 @@ impl RunSummaryStore {
.unwrap_or_else(PoisonError::into_inner) = Some(hook);
}
pub(crate) fn notify_platform_record(&self, run_id: RunId) {
/// Wake the run's projector: a platform record was committed for the run.
pub fn notify_platform_record(&self, run_id: RunId) {
let hook = self
.platform_hook
.read()

View file

@ -2039,6 +2039,45 @@ impl Client {
Ok(response.into_inner().hash)
}
/// Store one of Fabro's platform records for the run, tied to a Petri
/// stage when it belongs to one, and get it back as stored.
pub async fn append_petri_platform_record(
&self,
run_id: &RunId,
body: types::PetriPlatformRecordAppendRequest,
) -> Result<types::PetriPlatformRecord> {
let response = self
.send_api(|client| async move {
client
.append_petri_platform_record()
.id(run_id.to_string())
.body(body.clone())
.send()
.await
})
.await?;
Ok(response.into_inner())
}
/// The run's platform records in `seq` order, of one kind when `kind`
/// names it.
pub async fn list_petri_platform_records(
&self,
run_id: &RunId,
kind: Option<&str>,
) -> Result<Vec<types::PetriPlatformRecord>> {
let response = self
.send_api(|client| async move {
let mut request = client.list_petri_platform_records().id(run_id.to_string());
if let Some(kind) = kind {
request = request.kind(kind);
}
request.send().await
})
.await?;
Ok(response.into_inner().records)
}
/// The blob with this digest, or `None` when the store holds no such
/// blob. A run the store does not hold is an error.
pub async fn read_petri_blob(

View file

@ -38,6 +38,10 @@ impl EnvVars {
pub const FABRO_TEST_IN_MEMORY_STORE: &'static str = "FABRO_TEST_IN_MEMORY_STORE";
pub const FABRO_TEST_DISABLE_SPA_ASSETS: &'static str = "FABRO_TEST_DISABLE_SPA_ASSETS";
pub const FABRO_TEST_MODE: &'static str = "FABRO_TEST_MODE";
/// A directory of hold and release files a test uses to pause a Petri
/// run's checkpoint at a named point (`fabro_petri::hooks`); unset
/// outside tests.
pub const FABRO_TEST_CHECKPOINT_GATES: &'static str = "FABRO_TEST_CHECKPOINT_GATES";
pub const FABRO_VERBOSE: &'static str = "FABRO_VERBOSE";
pub const FABRO_WEB_URL: &'static str = "FABRO_WEB_URL";
pub const FABRO_WORKER_TOKEN: &'static str = "FABRO_WORKER_TOKEN";
@ -222,6 +226,7 @@ mod tests {
EnvVars::FABRO_TEST_IN_MEMORY_STORE,
EnvVars::FABRO_TEST_DISABLE_SPA_ASSETS,
EnvVars::FABRO_TEST_MODE,
EnvVars::FABRO_TEST_CHECKPOINT_GATES,
EnvVars::FABRO_VERBOSE,
EnvVars::FABRO_WEB_URL,
EnvVars::FABRO_WORKER_TOKEN,

View file

@ -299,6 +299,9 @@ models/petri-access.ts
models/petri-append-request.ts
models/petri-open-request.ts
models/petri-open-response.ts
models/petri-platform-record-append-request.ts
models/petri-platform-record-list.ts
models/petri-platform-record.ts
models/petri-record-list.ts
models/petri-record.ts
models/petri-release-request.ts

View file

@ -40,6 +40,12 @@ import type { PetriOpenRequest } from '../models';
// @ts-ignore
import type { PetriOpenResponse } from '../models';
// @ts-ignore
import type { PetriPlatformRecord } from '../models';
// @ts-ignore
import type { PetriPlatformRecordAppendRequest } from '../models';
// @ts-ignore
import type { PetriPlatformRecordList } from '../models';
// @ts-ignore
import type { PetriRecordList } from '../models';
// @ts-ignore
import type { PetriReleaseRequest } from '../models';
@ -66,6 +72,51 @@ import type { WriteRunBlobRequest } from '../models';
*/
export const RunInternalsApiAxiosParamCreator = function (configuration?: Configuration) {
return {
/**
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
* @summary Append Petri Platform Record
* @param {string} id Unique run identifier (ULID).
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
appendPetriPlatformRecord: async (id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options: RawAxiosRequestConfig = {}): Promise<RequestArgs> => {
// verify required parameter 'id' is not null or undefined
assertParamExists('appendPetriPlatformRecord', 'id', id)
// verify required parameter 'petriPlatformRecordAppendRequest' is not null or undefined
assertParamExists('appendPetriPlatformRecord', 'petriPlatformRecordAppendRequest', petriPlatformRecordAppendRequest)
const localVarPath = `/api/v1/runs/{id}/petri/platform-records`
.replace(`{${"id"}}`, encodeURIComponent(String(id)));
// use dummy base URL string because the URL constructor only accepts absolute URLs.
const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL);
let baseOptions;
if (configuration) {
baseOptions = configuration.baseOptions;
}
const localVarRequestOptions = { method: 'POST', ...baseOptions, ...options};
const localVarHeaderParameter = {} as any;
const localVarQueryParameter = {} as any;
// authentication SessionCookie required
// authentication BearerAuth required
// http bearer authentication required
await setBearerAuthToObject(localVarHeaderParameter, configuration)
localVarHeaderParameter['Content-Type'] = 'application/json';
localVarHeaderParameter['Accept'] = 'application/json';
setSearchParams(localVarUrlObj, localVarQueryParameter);
let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {};
localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers};
localVarRequestOptions.data = serializeDataIfNeeded(petriPlatformRecordAppendRequest, localVarRequestOptions, configuration)
return {
url: toPathString(localVarUrlObj),
options: localVarRequestOptions,
};
},
/**
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
* @summary Append Petri Records
@ -530,6 +581,51 @@ export const RunInternalsApiAxiosParamCreator = function (configuration?: Config
options: localVarRequestOptions,
};
},
/**
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
* @summary List Petri Platform Records
* @param {string} id Unique run identifier (ULID).
* @param {string} [kind] Only the platform records of this kind, as its &#x60;kind&#x60; tag spells it (&#x60;checkpoint&#x60;, &#x60;pull_request.created&#x60;, ...).
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
listPetriPlatformRecords: async (id: string, kind?: string, options: RawAxiosRequestConfig = {}): Promise<RequestArgs> => {
// verify required parameter 'id' is not null or undefined
assertParamExists('listPetriPlatformRecords', 'id', id)
const localVarPath = `/api/v1/runs/{id}/petri/platform-records`
.replace(`{${"id"}}`, encodeURIComponent(String(id)));
// use dummy base URL string because the URL constructor only accepts absolute URLs.
const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL);
let baseOptions;
if (configuration) {
baseOptions = configuration.baseOptions;
}
const localVarRequestOptions = { method: 'GET', ...baseOptions, ...options};
const localVarHeaderParameter = {} as any;
const localVarQueryParameter = {} as any;
// authentication SessionCookie required
// authentication BearerAuth required
// http bearer authentication required
await setBearerAuthToObject(localVarHeaderParameter, configuration)
if (kind !== undefined) {
localVarQueryParameter['kind'] = kind;
}
localVarHeaderParameter['Accept'] = 'application/json';
setSearchParams(localVarUrlObj, localVarQueryParameter);
let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {};
localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers};
return {
url: toPathString(localVarUrlObj),
options: localVarRequestOptions,
};
},
/**
* Every record of one log of the run, in `seq` order, unchanged.
* @summary List Petri Records
@ -1246,6 +1342,20 @@ export const RunInternalsApiAxiosParamCreator = function (configuration?: Config
export const RunInternalsApiFp = function(configuration?: Configuration) {
const localVarAxiosParamCreator = RunInternalsApiAxiosParamCreator(configuration)
return {
/**
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
* @summary Append Petri Platform Record
* @param {string} id Unique run identifier (ULID).
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
async appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise<PetriPlatformRecord>> {
const localVarAxiosArgs = await localVarAxiosParamCreator.appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options);
const localVarOperationServerIndex = configuration?.serverIndex ?? 0;
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.appendPetriPlatformRecord']?.[localVarOperationServerIndex]?.url;
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
},
/**
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
* @summary Append Petri Records
@ -1389,6 +1499,20 @@ export const RunInternalsApiFp = function(configuration?: Configuration) {
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.getStageArtifact']?.[localVarOperationServerIndex]?.url;
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
},
/**
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
* @summary List Petri Platform Records
* @param {string} id Unique run identifier (ULID).
* @param {string} [kind] Only the platform records of this kind, as its &#x60;kind&#x60; tag spells it (&#x60;checkpoint&#x60;, &#x60;pull_request.created&#x60;, ...).
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
async listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise<PetriPlatformRecordList>> {
const localVarAxiosArgs = await localVarAxiosParamCreator.listPetriPlatformRecords(id, kind, options);
const localVarOperationServerIndex = configuration?.serverIndex ?? 0;
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.listPetriPlatformRecords']?.[localVarOperationServerIndex]?.url;
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
},
/**
* Every record of one log of the run, in `seq` order, unchanged.
* @summary List Petri Records
@ -1615,6 +1739,17 @@ export const RunInternalsApiFp = function(configuration?: Configuration) {
export const RunInternalsApiFactory = function (configuration?: Configuration, basePath?: string, axios?: AxiosInstance) {
const localVarFp = RunInternalsApiFp(configuration)
return {
/**
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
* @summary Append Petri Platform Record
* @param {string} id Unique run identifier (ULID).
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig): AxiosPromise<PetriPlatformRecord> {
return localVarFp.appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options).then((request) => request(axios, basePath));
},
/**
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
* @summary Append Petri Records
@ -1728,6 +1863,17 @@ export const RunInternalsApiFactory = function (configuration?: Configuration, b
getStageArtifact(id: string, stageId: string, filename: string, retry: number, options?: RawAxiosRequestConfig): AxiosPromise<File> {
return localVarFp.getStageArtifact(id, stageId, filename, retry, options).then((request) => request(axios, basePath));
},
/**
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
* @summary List Petri Platform Records
* @param {string} id Unique run identifier (ULID).
* @param {string} [kind] Only the platform records of this kind, as its &#x60;kind&#x60; tag spells it (&#x60;checkpoint&#x60;, &#x60;pull_request.created&#x60;, ...).
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig): AxiosPromise<PetriPlatformRecordList> {
return localVarFp.listPetriPlatformRecords(id, kind, options).then((request) => request(axios, basePath));
},
/**
* Every record of one log of the run, in `seq` order, unchanged.
* @summary List Petri Records
@ -1907,6 +2053,18 @@ export const RunInternalsApiFactory = function (configuration?: Configuration, b
* RunInternalsApi - object-oriented interface
*/
export class RunInternalsApi extends BaseAPI {
/**
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
* @summary Append Petri Platform Record
* @param {string} id Unique run identifier (ULID).
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
public appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig) {
return RunInternalsApiFp(this.configuration).appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options).then((request) => request(this.axios, this.basePath));
}
/**
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
* @summary Append Petri Records
@ -2030,6 +2188,18 @@ export class RunInternalsApi extends BaseAPI {
return RunInternalsApiFp(this.configuration).getStageArtifact(id, stageId, filename, retry, options).then((request) => request(this.axios, this.basePath));
}
/**
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
* @summary List Petri Platform Records
* @param {string} id Unique run identifier (ULID).
* @param {string} [kind] Only the platform records of this kind, as its &#x60;kind&#x60; tag spells it (&#x60;checkpoint&#x60;, &#x60;pull_request.created&#x60;, ...).
* @param {*} [options] Override http request option.
* @throws {RequiredError}
*/
public listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig) {
return RunInternalsApiFp(this.configuration).listPetriPlatformRecords(id, kind, options).then((request) => request(this.axios, this.basePath));
}
/**
* Every record of one log of the run, in `seq` order, unchanged.
* @summary List Petri Records

View file

@ -269,6 +269,9 @@ export * from './petri-access';
export * from './petri-append-request';
export * from './petri-open-request';
export * from './petri-open-response';
export * from './petri-platform-record';
export * from './petri-platform-record-append-request';
export * from './petri-platform-record-list';
export * from './petri-record';
export * from './petri-record-list';
export * from './petri-release-request';

View file

@ -0,0 +1,33 @@
/* tslint:disable */
/* eslint-disable */
/**
* Fabro Run API
* HTTP API for managing Fabro workflow run executions.
*
* The version of the OpenAPI document: 0.2.0
*
*
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
* https://openapi-generator.tech
* Do not edit the class manually.
*/
/**
* One platform record to store for the run.
*/
export interface PetriPlatformRecordAppendRequest {
/**
* The platform record, tagged by `kind`.
*/
'record': { [key: string]: any; };
/**
* The Petri execution the record belongs to, with `firing`.
*/
'execution'?: number;
/**
* The Petri firing the record belongs to, with `execution`.
*/
'firing'?: number;
}

View file

@ -0,0 +1,25 @@
/* tslint:disable */
/* eslint-disable */
/**
* Fabro Run API
* HTTP API for managing Fabro workflow run executions.
*
* The version of the OpenAPI document: 0.2.0
*
*
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
* https://openapi-generator.tech
* Do not edit the class manually.
*/
// May contain unused imports in some cases
// @ts-ignore
import type { PetriPlatformRecord } from './petri-platform-record';
/**
* The run\'s platform records, in `seq` order.
*/
export interface PetriPlatformRecordList {
'records': Array<PetriPlatformRecord>;
}

View file

@ -0,0 +1,41 @@
/* tslint:disable */
/* eslint-disable */
/**
* Fabro Run API
* HTTP API for managing Fabro workflow run executions.
*
* The version of the OpenAPI document: 0.2.0
*
*
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
* https://openapi-generator.tech
* Do not edit the class manually.
*/
/**
* One of Fabro\'s platform records of a Petri run, as stored: the record\'s JSON tagged by `kind`, its position in the run\'s platform record sequence, and the Petri stage it belongs to when it belongs to one.
*/
export interface PetriPlatformRecord {
/**
* The record\'s position in the run\'s platform records, from 1.
*/
'seq': number;
/**
* Milliseconds since the Unix epoch when the record was stored.
*/
'recorded_at': number;
/**
* The platform record itself, tagged by `kind`.
*/
'record': { [key: string]: any; };
/**
* The Petri execution the record belongs to, with `firing`.
*/
'execution'?: number;
/**
* The Petri firing the record belongs to, with `execution`.
*/
'firing'?: number;
}