mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-01 02:04:24 +00:00
Merge branch 'petri-integration-hooks' into petri-integration
# Conflicts: # Cargo.lock # lib/apps/fabro-cli/src/commands/run/petri_worker.rs # lib/apps/fabro-cli/tests/it/scenario/petri.rs # lib/apps/fabro-server/src/server/petri_runs.rs # lib/components/fabro-petri/Cargo.toml # lib/components/fabro-petri/src/engine.rs # lib/components/fabro-petri/src/lib.rs # lib/components/fabro-store/src/platform_records.rs
This commit is contained in:
commit
c9c4c27753
29 changed files with 4329 additions and 43 deletions
2
Cargo.lock
generated
2
Cargo.lock
generated
|
|
@ -2894,11 +2894,13 @@ dependencies = [
|
|||
"chrono",
|
||||
"fabro-api",
|
||||
"fabro-auth",
|
||||
"fabro-checkpoint",
|
||||
"fabro-client",
|
||||
"fabro-db",
|
||||
"fabro-http",
|
||||
"fabro-interview",
|
||||
"fabro-llm",
|
||||
"fabro-petri",
|
||||
"fabro-store",
|
||||
"fabro-test",
|
||||
"fabro-types",
|
||||
|
|
|
|||
|
|
@ -3418,6 +3418,59 @@ paths:
|
|||
schema:
|
||||
$ref: "#/components/schemas/ErrorResponse"
|
||||
|
||||
/api/v1/runs/{id}/petri/platform-records:
|
||||
get:
|
||||
operationId: listPetriPlatformRecords
|
||||
tags: [Run Internals]
|
||||
summary: List Petri Platform Records
|
||||
description: |
|
||||
The run's platform records (Fabro's own facts about a Petri run: a
|
||||
checkpoint commit, a pull request, a notification), in `seq` order,
|
||||
optionally of one kind. What a run's worker reads to find an effect
|
||||
it already performed before performing it again.
|
||||
parameters:
|
||||
- $ref: "#/components/parameters/RunId"
|
||||
- $ref: "#/components/parameters/PetriPlatformRecordKind"
|
||||
responses:
|
||||
"200":
|
||||
description: The run's platform records
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: "#/components/schemas/PetriPlatformRecordList"
|
||||
post:
|
||||
operationId: appendPetriPlatformRecord
|
||||
tags: [Run Internals]
|
||||
summary: Append Petri Platform Record
|
||||
description: |
|
||||
Stores one platform record at the run's next `seq`, tied to the
|
||||
Petri stage named by `execution` and `firing` when it belongs to one.
|
||||
The record is the JSON of a Fabro platform record, tagged by `kind`.
|
||||
parameters:
|
||||
- $ref: "#/components/parameters/RunId"
|
||||
requestBody:
|
||||
required: true
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: "#/components/schemas/PetriPlatformRecordAppendRequest"
|
||||
responses:
|
||||
"200":
|
||||
description: The record as stored
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: "#/components/schemas/PetriPlatformRecord"
|
||||
"400":
|
||||
description: The record is not a platform record
|
||||
headers:
|
||||
x-request-id:
|
||||
$ref: "#/components/headers/XRequestId"
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: "#/components/schemas/ErrorResponse"
|
||||
|
||||
/api/v1/runs/{id}/stages/{stageId}/logs/output:
|
||||
get:
|
||||
operationId: getRunStageCommandLog
|
||||
|
|
@ -6222,6 +6275,17 @@ components:
|
|||
type: string
|
||||
example: 18f3c2a9e1b4-42017-0-9f3a1c7e2b5d
|
||||
|
||||
PetriPlatformRecordKind:
|
||||
name: kind
|
||||
in: query
|
||||
required: false
|
||||
description: >-
|
||||
Only the platform records of this kind, as its `kind` tag spells it
|
||||
(`checkpoint`, `pull_request.created`, ...).
|
||||
schema:
|
||||
type: string
|
||||
example: checkpoint
|
||||
|
||||
ArtifactFilename:
|
||||
name: filename
|
||||
in: query
|
||||
|
|
@ -11000,6 +11064,71 @@ components:
|
|||
items:
|
||||
$ref: "#/components/schemas/PetriRecord"
|
||||
|
||||
PetriPlatformRecord:
|
||||
description: >-
|
||||
One of Fabro's platform records of a Petri run, as stored: the
|
||||
record's JSON tagged by `kind`, its position in the run's platform
|
||||
record sequence, and the Petri stage it belongs to when it belongs
|
||||
to one.
|
||||
type: object
|
||||
required:
|
||||
- seq
|
||||
- recorded_at
|
||||
- record
|
||||
properties:
|
||||
seq:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: The record's position in the run's platform records, from 1.
|
||||
example: 4
|
||||
recorded_at:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: Milliseconds since the Unix epoch when the record was stored.
|
||||
example: 1758067200123
|
||||
record:
|
||||
type: object
|
||||
additionalProperties: true
|
||||
description: The platform record itself, tagged by `kind`.
|
||||
execution:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: The Petri execution the record belongs to, with `firing`.
|
||||
firing:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: The Petri firing the record belongs to, with `execution`.
|
||||
|
||||
PetriPlatformRecordAppendRequest:
|
||||
description: One platform record to store for the run.
|
||||
type: object
|
||||
required:
|
||||
- record
|
||||
properties:
|
||||
record:
|
||||
type: object
|
||||
additionalProperties: true
|
||||
description: The platform record, tagged by `kind`.
|
||||
execution:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: The Petri execution the record belongs to, with `firing`.
|
||||
firing:
|
||||
type: integer
|
||||
format: uint64
|
||||
description: The Petri firing the record belongs to, with `execution`.
|
||||
|
||||
PetriPlatformRecordList:
|
||||
description: The run's platform records, in `seq` order.
|
||||
type: object
|
||||
required:
|
||||
- records
|
||||
properties:
|
||||
records:
|
||||
type: array
|
||||
items:
|
||||
$ref: "#/components/schemas/PetriPlatformRecord"
|
||||
|
||||
CommandTermination:
|
||||
description: Terminal state for a command execution.
|
||||
type: string
|
||||
|
|
|
|||
|
|
@ -27,6 +27,10 @@
|
|||
//! for good cancels the run the same way, and the worker exits with that
|
||||
//! loss as its error once the run has settled.
|
||||
//!
|
||||
//! Fabro's hooks ride the run with their platform records over the same
|
||||
//! client: the checkpoint commit in the run's host workspace before every
|
||||
//! durable finish, and its record after every route.
|
||||
//!
|
||||
//! The runtime's settings layer is left empty here: the run's graphs were
|
||||
//! lowered and admitted at create time with the server's layer, and nothing
|
||||
//! lowers again at execution. The model client is built from the worker's
|
||||
|
|
@ -47,11 +51,14 @@ use fabro_interview::ControlInterviewer;
|
|||
use fabro_llm::credentials::{CredentialProvider, readiness};
|
||||
use fabro_petri::blobs::ClientBlobs;
|
||||
use fabro_petri::engine::{self, Conclusion, Execution, RunRequest};
|
||||
use fabro_petri::hooks::HooksSpec;
|
||||
use fabro_petri::interview::{Approval, EventSinkQuestions, FabroInterviewer};
|
||||
use fabro_petri::petri::OwnerId;
|
||||
use fabro_petri::platform_records::HttpPlatformRecords;
|
||||
use fabro_petri::runtime::{self, RuntimeSpec};
|
||||
use fabro_petri::secrets::VaultSecrets;
|
||||
use fabro_petri::{HttpRunStore, admission};
|
||||
use fabro_static::EnvVars;
|
||||
use fabro_store::RunProjection;
|
||||
use fabro_types::settings::run::{ApprovalMode, RunMode};
|
||||
use fabro_types::{FailureReason, RunId, RunTiming, StageOutcome, SuccessReason};
|
||||
|
|
@ -166,6 +173,11 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
|
|||
}
|
||||
runner::set_worker_title(&run_id, WorkerTitlePhase::Running);
|
||||
|
||||
let hooks = HooksSpec::for_run(
|
||||
Arc::new(HttpPlatformRecords::new(worker.client.clone_for_reuse())),
|
||||
&worker.run_state.spec.settings.run,
|
||||
)
|
||||
.with_test_gates(test_checkpoint_gates());
|
||||
let request = RunRequest {
|
||||
run_id: run_id.to_string(),
|
||||
run_dir: worker.run_dir.join("petri"),
|
||||
|
|
@ -188,6 +200,7 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
|
|||
worker.client.clone_for_reuse(),
|
||||
run_id,
|
||||
))),
|
||||
hooks: Some(hooks),
|
||||
};
|
||||
let run = Box::pin(engine::run(request));
|
||||
tokio::pin!(run);
|
||||
|
|
@ -259,6 +272,15 @@ pub(super) async fn execute(worker: PetriWorker<'_>) -> Result<()> {
|
|||
}
|
||||
}
|
||||
|
||||
/// A test's checkpoint gate directory, when the server forwarded one.
|
||||
#[expect(
|
||||
clippy::disallowed_methods,
|
||||
reason = "the gate directory is a test-only process-env facade the server forwards by name"
|
||||
)]
|
||||
fn test_checkpoint_gates() -> Option<PathBuf> {
|
||||
std::env::var_os(EnvVars::FABRO_TEST_CHECKPOINT_GATES).map(PathBuf::from)
|
||||
}
|
||||
|
||||
/// The runtime the worker hands Petri: no settings layer (nothing lowers
|
||||
/// at execution), the model client over the worker's catalog and vault for
|
||||
/// the providers whose credentials resolve, the run's mode, and the Fabro
|
||||
|
|
|
|||
|
|
@ -29,12 +29,13 @@ use std::time::{Duration, Instant};
|
|||
use fabro_client::ServerTarget;
|
||||
use fabro_config::{Storage, envfile};
|
||||
use fabro_petri::SqliteRunStore;
|
||||
use fabro_petri::checkpoint::CheckpointKey;
|
||||
use fabro_petri::engine::{self, RunStatus};
|
||||
use fabro_petri::petri::RunKey;
|
||||
use fabro_static::EnvVars;
|
||||
use fabro_store::EventEnvelope;
|
||||
use fabro_store::{EventEnvelope, PlatformRecord, PlatformRecordKind, PlatformRecordStore};
|
||||
use fabro_test::{apply_test_isolation, expect_reqwest_json, isolated_storage_dir, test_context};
|
||||
use fabro_types::EventBody;
|
||||
use fabro_types::{EventBody, RunId};
|
||||
|
||||
use crate::cmd::support::created_run_id;
|
||||
use crate::support::{
|
||||
|
|
@ -81,6 +82,8 @@ struct RunningServer {
|
|||
config_path: PathBuf,
|
||||
port: u16,
|
||||
api_base_url: String,
|
||||
/// The checkpoint gate directory the server forwards to its workers.
|
||||
gates_dir: PathBuf,
|
||||
}
|
||||
|
||||
impl RunningServer {
|
||||
|
|
@ -103,6 +106,8 @@ impl RunningServer {
|
|||
.expect("the server env writes");
|
||||
fabro_util::dev_token::write_dev_token(&runtime_directory.dev_token_path(), TEST_DEV_TOKEN)
|
||||
.expect("the dev token writes");
|
||||
let gates_dir = home_root.path().join("checkpoint-gates");
|
||||
std::fs::create_dir_all(&gates_dir).expect("the gates dir creates");
|
||||
let mut server = Self {
|
||||
child: None,
|
||||
home_root,
|
||||
|
|
@ -111,6 +116,7 @@ impl RunningServer {
|
|||
config_path,
|
||||
port,
|
||||
api_base_url: format!("http://127.0.0.1:{port}"),
|
||||
gates_dir,
|
||||
};
|
||||
server.launch().await;
|
||||
server
|
||||
|
|
@ -129,6 +135,7 @@ impl RunningServer {
|
|||
EnvVars::FABRO_HOME,
|
||||
self.home_root.path().join("fabro-home"),
|
||||
);
|
||||
cmd.env(EnvVars::FABRO_TEST_CHECKPOINT_GATES, &self.gates_dir);
|
||||
cmd.args(["server", "start", "--foreground"])
|
||||
.arg("--storage-dir")
|
||||
.arg(&self.storage_dir)
|
||||
|
|
@ -137,16 +144,16 @@ impl RunningServer {
|
|||
.arg("--config")
|
||||
.arg(&self.config_path)
|
||||
.stdin(Stdio::null())
|
||||
.stdout(Stdio::null())
|
||||
.stdout(self.stderr_log())
|
||||
.stderr(self.stderr_log());
|
||||
let mut child = cmd.spawn().expect("the server spawns");
|
||||
wait_for_http_ready(&self.api_base_url, &mut child).await;
|
||||
self.child = Some(child);
|
||||
}
|
||||
|
||||
/// Where the server's stderr goes: a file beside its storage, so a
|
||||
/// chatty server never blocks on a pipe nobody reads, and a failing
|
||||
/// test can show it.
|
||||
/// Where the server's stdout and stderr go: a file beside its storage,
|
||||
/// so a chatty server never blocks on a pipe nobody reads, and a
|
||||
/// failing test can show its log.
|
||||
fn stderr_log(&self) -> Stdio {
|
||||
let path = self.storage_dir.with_file_name("server.stderr.log");
|
||||
let file = std::fs::OpenOptions::new()
|
||||
|
|
@ -209,6 +216,117 @@ impl RunningServer {
|
|||
.expect("the server database opens");
|
||||
SqliteRunStore::new(database.clone_pool())
|
||||
}
|
||||
|
||||
/// The run's platform records in the server's database.
|
||||
async fn platform_records(&self) -> PlatformRecordStore {
|
||||
let database = fabro_db::Database::connect(Storage::new(&self.storage_dir).sqlite_path())
|
||||
.await
|
||||
.expect("the server database opens");
|
||||
PlatformRecordStore::new(database.clone_pool())
|
||||
}
|
||||
|
||||
/// Where the run's worker ran Petri: the run's scratch under the
|
||||
/// server's storage.
|
||||
fn petri_run_dir(&self, run_id: &str) -> PathBuf {
|
||||
let run_id: RunId = run_id.parse().expect("the run id parses");
|
||||
Storage::new(&self.storage_dir)
|
||||
.run_scratch(&run_id)
|
||||
.root()
|
||||
.join("petri")
|
||||
}
|
||||
|
||||
/// The worker's own log for the run.
|
||||
fn worker_log(&self, run_id: &str) -> PathBuf {
|
||||
let run_id: RunId = run_id.parse().expect("the run id parses");
|
||||
Storage::new(&self.storage_dir)
|
||||
.run_scratch(&run_id)
|
||||
.root()
|
||||
.join("runtime")
|
||||
.join("server.log")
|
||||
}
|
||||
|
||||
/// Hold the worker's checkpoint at `point` (`commit` or `record`) for
|
||||
/// `node` until [`release`](Self::release).
|
||||
fn hold(&self, point: &str, node: &str) {
|
||||
std::fs::write(self.gates_dir.join(format!("{point}.{node}.hold")), "")
|
||||
.expect("the hold file writes");
|
||||
}
|
||||
|
||||
fn release(&self, point: &str, node: &str) {
|
||||
std::fs::write(self.gates_dir.join(format!("{point}.{node}.release")), "")
|
||||
.expect("the release file writes");
|
||||
}
|
||||
|
||||
/// Wait until the worker's log says its checkpoint is held at a gate.
|
||||
fn wait_until_held(&self, run_id: &str, point: &str, node: &str) {
|
||||
let log = self.worker_log(run_id);
|
||||
let needle = format!("checkpoint held at a test gate point=\"{point}\" node=\"{node}\"");
|
||||
let deadline = Instant::now() + RUN_TIMEOUT;
|
||||
loop {
|
||||
let text = std::fs::read_to_string(&log).unwrap_or_default();
|
||||
if text.contains(&needle) {
|
||||
return;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"the worker never held at {point}.{node}; log:\n{text}"
|
||||
);
|
||||
std::thread::sleep(POLL);
|
||||
}
|
||||
}
|
||||
|
||||
/// The one host workspace of the run, and the commits on its run
|
||||
/// branch, oldest first, as `(subject, key)`.
|
||||
fn workspace_commits(&self, run_id: &str) -> (PathBuf, Vec<(String, Option<CheckpointKey>)>) {
|
||||
let scopes = self.petri_run_dir(run_id).join("scopes");
|
||||
let mut workspaces: Vec<PathBuf> = std::fs::read_dir(&scopes)
|
||||
.expect("the scopes directory lists")
|
||||
.map(|entry| entry.expect("an entry reads").path().join("work"))
|
||||
.collect();
|
||||
assert_eq!(workspaces.len(), 1, "one workspace: {workspaces:?}");
|
||||
let workspace = workspaces.remove(0);
|
||||
let output = Command::new("git")
|
||||
.args(["log", "--reverse", "--format=%s%x00%B%x1e"])
|
||||
.current_dir(&workspace)
|
||||
.output()
|
||||
.expect("git runs");
|
||||
assert!(
|
||||
output.status.success(),
|
||||
"git log failed: {}",
|
||||
String::from_utf8_lossy(&output.stderr)
|
||||
);
|
||||
let log = String::from_utf8_lossy(&output.stdout).into_owned();
|
||||
let commits = log
|
||||
.split('\u{1e}')
|
||||
.filter(|entry| !entry.trim().is_empty())
|
||||
.map(|entry| {
|
||||
let mut parts = entry.trim_start().splitn(2, '\0');
|
||||
let subject = parts.next().unwrap_or_default().to_string();
|
||||
let body = parts.next().unwrap_or_default();
|
||||
(subject, CheckpointKey::from_message(body))
|
||||
})
|
||||
.collect();
|
||||
(workspace, commits)
|
||||
}
|
||||
|
||||
/// The run's checkpoint records, in seq order, as `(node position, sha)`.
|
||||
async fn checkpoints(&self, run_id: &str) -> Vec<(CheckpointKey, String)> {
|
||||
let run_id: RunId = run_id.parse().expect("the run id parses");
|
||||
self.platform_records()
|
||||
.await
|
||||
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.expect("the checkpoint records read")
|
||||
.into_iter()
|
||||
.filter_map(|stored| match stored.record {
|
||||
PlatformRecord::Checkpoint(record) => Some((
|
||||
CheckpointKey::from_operation(record.operation.as_ref()?)?,
|
||||
record.git_commit_sha?,
|
||||
)),
|
||||
_ => None,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for RunningServer {
|
||||
|
|
@ -357,6 +475,28 @@ async fn run_events(server: &RunningServer, run_id: &str) -> Vec<EventEnvelope>
|
|||
parse_event_envelopes(&run_json(server, &format!("runs/{run_id}/events")).await)
|
||||
}
|
||||
|
||||
/// The run's events once `terminal` is among them: the events list is a
|
||||
/// projection that settles after the run's status does.
|
||||
async fn settled_events(
|
||||
server: &RunningServer,
|
||||
run_id: &str,
|
||||
terminal: &str,
|
||||
) -> Vec<EventEnvelope> {
|
||||
let deadline = Instant::now() + RUN_TIMEOUT;
|
||||
loop {
|
||||
let events = run_events(server, run_id).await;
|
||||
if event_names(&events).contains(&terminal) {
|
||||
return events;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"`{terminal}` was never projected for run {run_id}: {:?}",
|
||||
event_names(&events)
|
||||
);
|
||||
tokio::time::sleep(POLL).await;
|
||||
}
|
||||
}
|
||||
|
||||
fn event_names(events: &[EventEnvelope]) -> Vec<&str> {
|
||||
events
|
||||
.iter()
|
||||
|
|
@ -432,7 +572,7 @@ async fn a_petri_run_executes_in_the_server_launched_worker() {
|
|||
let state = run_json(&server, &format!("runs/{run_id}/state")).await;
|
||||
assert_eq!(state["spec"]["engine"]["kind"], "petri", "state: {state}");
|
||||
|
||||
let events = run_events(&server, &run_id).await;
|
||||
let events = settled_events(&server, &run_id, "run.completed").await;
|
||||
let names = event_names(&events);
|
||||
assert_eq!(
|
||||
names
|
||||
|
|
@ -519,14 +659,14 @@ async fn a_petri_run_resumes_in_a_new_worker_after_the_server_restarts() {
|
|||
std::fs::write(&gate, "go").expect("the gate opens");
|
||||
|
||||
let status = wait_for_status(&server, &run_id, &["succeeded", "failed"]).await;
|
||||
let events = run_events(&server, &run_id).await;
|
||||
let names = event_names(&events);
|
||||
assert_eq!(
|
||||
status,
|
||||
"succeeded",
|
||||
"events: {names:?}\nserver stderr:\n{}",
|
||||
"server stderr:\n{}",
|
||||
server.stderr_text()
|
||||
);
|
||||
let events = settled_events(&server, &run_id, "run.completed").await;
|
||||
let names = event_names(&events);
|
||||
assert_eq!(
|
||||
names
|
||||
.iter()
|
||||
|
|
@ -811,3 +951,375 @@ async fn an_unanswered_gate_in_the_worker_expires_with_its_default() {
|
|||
assert!(questions(&server, &run_id).await.is_empty());
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A shell loop that waits for `gate` to exist.
|
||||
fn wait_for(gate: &Path) -> String {
|
||||
format!("while [ ! -f {} ]; do sleep 0.05; done", gate.display())
|
||||
}
|
||||
|
||||
/// Three command stages: `one` writes a file, `two` writes another after
|
||||
/// the gate opens (and logs each run), `three` checks both files.
|
||||
fn three_stage_bundle(context: &fabro_test::TestContext, gate: &Path) -> PathBuf {
|
||||
write_petri_workflow(
|
||||
context,
|
||||
&format!(
|
||||
"digraph Stages {{\n graph [goal=\"Three stages\", default_max_retries=0]\n start \
|
||||
[shape=Mdiamond]\n exit [shape=Msquare]\n one [shape=parallelogram, script=\"echo \
|
||||
one > one.txt\"]\n two [shape=parallelogram, script=\"echo run >> two.log; {}; \
|
||||
echo two > two.txt\"]\n three [shape=parallelogram, script=\"test \\\"$(cat \
|
||||
one.txt)\\\" = one && test \\\"$(cat two.txt)\\\" = two && cp two.log \
|
||||
three.log\"]\n start -> one -> two -> three -> exit\n}}\n",
|
||||
wait_for(gate)
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
/// Kill the server first, so it never observes the worker exit, then the
|
||||
/// worker's whole process group, then the stage's own process group when a
|
||||
/// stage was waiting on `gate`: a stage process runs in a group of its own
|
||||
/// under the host plugin, and a machine crash takes it with everything
|
||||
/// else, where a killed worker alone would leave it writing into the
|
||||
/// workspace.
|
||||
fn crash(server: &mut RunningServer, worker: u32, gate: Option<&Path>) {
|
||||
server.kill();
|
||||
fabro_proc::sigkill_process_group(worker);
|
||||
let deadline = Instant::now() + Duration::from_secs(10);
|
||||
while fabro_proc::process_running(worker) {
|
||||
assert!(Instant::now() < deadline, "the worker did not die");
|
||||
std::thread::sleep(POLL);
|
||||
}
|
||||
let Some(gate) = gate else {
|
||||
return;
|
||||
};
|
||||
let output = Command::new("pgrep")
|
||||
.args(["-f", &gate.display().to_string()])
|
||||
.output()
|
||||
.expect("pgrep runs");
|
||||
for pid in String::from_utf8_lossy(&output.stdout)
|
||||
.lines()
|
||||
.filter_map(|line| line.trim().parse::<u32>().ok())
|
||||
{
|
||||
fabro_proc::sigkill_process_group(pid);
|
||||
}
|
||||
let deadline = Instant::now() + Duration::from_secs(10);
|
||||
while gate_is_polled(gate) {
|
||||
assert!(Instant::now() < deadline, "the stage did not die");
|
||||
std::thread::sleep(POLL);
|
||||
}
|
||||
}
|
||||
|
||||
/// Wait for the run to succeed after a restart, with the server's stderr
|
||||
/// on failure.
|
||||
async fn wait_for_success(server: &RunningServer, run_id: &str) {
|
||||
let status = wait_for_status(server, run_id, &["succeeded", "failed"]).await;
|
||||
assert_eq!(
|
||||
status,
|
||||
"succeeded",
|
||||
"run: {}\nserver stderr:\n{}",
|
||||
run_json(server, &format!("runs/{run_id}")).await,
|
||||
server.stderr_text()
|
||||
);
|
||||
}
|
||||
|
||||
/// The subjects of the commits on the run branch.
|
||||
fn subjects(commits: &[(String, Option<CheckpointKey>)]) -> Vec<&str> {
|
||||
commits
|
||||
.iter()
|
||||
.map(|(subject, _)| subject.as_str())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The commit subjects one run of the three-stage bundle produces.
|
||||
fn three_stage_subjects(run_id: &str) -> Vec<String> {
|
||||
["start", "one", "two", "three", "exit"]
|
||||
.iter()
|
||||
.map(|node| format!("fabro({run_id}): {node} (success)"))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// A worker killed after a stage's finish is durable: on the restart the
|
||||
/// stage's commit is not repeated, the stage in flight reruns on the
|
||||
/// snapshot (its partial output gone), and the next stage sees both.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_crash_after_a_durable_finish_keeps_its_one_commit() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let gate = context.temp_dir.join("two.gate");
|
||||
let workspace = three_stage_bundle(&context, &gate);
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
|
||||
wait_for_status(&server, &run_id, &["running"]).await;
|
||||
let worker = wait_for_worker(&run_id);
|
||||
wait_until_gate_is_polled(&gate);
|
||||
crash(&mut server, worker, Some(&gate));
|
||||
|
||||
server.launch().await;
|
||||
let resumed = wait_for_worker(&run_id);
|
||||
assert_ne!(resumed, worker);
|
||||
wait_until_gate_is_polled(&gate);
|
||||
std::fs::write(&gate, "go").expect("the gate opens");
|
||||
wait_for_success(&server, &run_id).await;
|
||||
|
||||
let (path, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
|
||||
"run\n",
|
||||
"the crashed attempt's partial output was reset before the rerun; server log:\n{}",
|
||||
server.stderr_text()
|
||||
);
|
||||
let checkpoints = server.checkpoints(&run_id).await;
|
||||
assert_eq!(checkpoints.len(), 5, "{checkpoints:?}");
|
||||
let keys: Vec<Option<CheckpointKey>> = checkpoints.iter().map(|(key, _)| Some(*key)).collect();
|
||||
let committed: Vec<Option<CheckpointKey>> = commits.iter().map(|(_, key)| *key).collect();
|
||||
assert_eq!(keys, committed);
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A worker killed in `prepare_result` before the commit lands: the finish
|
||||
/// is not durable, the stage reruns once, and one commit exists for it.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_crash_before_the_commit_lands_reruns_the_stage_once() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let gate = context.temp_dir.join("two.gate");
|
||||
std::fs::write(&gate, "open").expect("the script gate is open from the start");
|
||||
let workspace = three_stage_bundle(&context, &gate);
|
||||
server.hold("commit", "two");
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
|
||||
wait_for_status(&server, &run_id, &["running"]).await;
|
||||
let worker = wait_for_worker(&run_id);
|
||||
server.wait_until_held(&run_id, "commit", "two");
|
||||
crash(&mut server, worker, None);
|
||||
|
||||
server.release("commit", "two");
|
||||
server.launch().await;
|
||||
wait_for_success(&server, &run_id).await;
|
||||
|
||||
let (path, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
|
||||
"run\n",
|
||||
"the stage reran once, on the snapshot before it"
|
||||
);
|
||||
let store = server.petri_store().await;
|
||||
let outcome = engine::outcome_of(&store, &run_id)
|
||||
.await
|
||||
.expect("the run's Petri record inspects");
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
assert!(outcome.complete, "{:?}", outcome.incomplete);
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A worker killed after the commit and its durable finish but before the
|
||||
/// platform record: the restart reconciles the record from the snapshot
|
||||
/// repository, the stage does not rerun, and one commit exists for it.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_crash_before_the_record_reconciles_it_from_the_run_branch() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let gate = context.temp_dir.join("two.gate");
|
||||
std::fs::write(&gate, "open").expect("the script gate is open from the start");
|
||||
let workspace = three_stage_bundle(&context, &gate);
|
||||
server.hold("record", "two");
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
|
||||
wait_for_status(&server, &run_id, &["running"]).await;
|
||||
let worker = wait_for_worker(&run_id);
|
||||
server.wait_until_held(&run_id, "record", "two");
|
||||
let before = server.checkpoints(&run_id).await;
|
||||
assert_eq!(before.len(), 2, "start and one are recorded: {before:?}");
|
||||
crash(&mut server, worker, None);
|
||||
|
||||
server.release("record", "two");
|
||||
server.launch().await;
|
||||
wait_for_success(&server, &run_id).await;
|
||||
|
||||
let (path, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(path.join("three.log")).expect("three copied the log"),
|
||||
"run\n",
|
||||
"the stage with a durable finish did not rerun"
|
||||
);
|
||||
let checkpoints = server.checkpoints(&run_id).await;
|
||||
assert_eq!(checkpoints.len(), 5, "{checkpoints:?}");
|
||||
let keys: Vec<Option<CheckpointKey>> = checkpoints.iter().map(|(key, _)| Some(*key)).collect();
|
||||
let committed: Vec<Option<CheckpointKey>> = commits.iter().map(|(_, key)| *key).collect();
|
||||
assert_eq!(
|
||||
keys, committed,
|
||||
"the reconciled record names the one commit"
|
||||
);
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A workspace deleted while the run is down is restored from the snapshot
|
||||
/// repository, and the next stage sees the checkpoint's files.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_deleted_workspace_is_restored_from_its_snapshot() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let gate = context.temp_dir.join("two.gate");
|
||||
let workspace = three_stage_bundle(&context, &gate);
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
|
||||
wait_for_status(&server, &run_id, &["running"]).await;
|
||||
let worker = wait_for_worker(&run_id);
|
||||
wait_until_gate_is_polled(&gate);
|
||||
crash(&mut server, worker, Some(&gate));
|
||||
let (path, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(subjects(&commits), three_stage_subjects(&run_id)[..2]);
|
||||
std::fs::remove_dir_all(&path).expect("the workspace is deleted");
|
||||
|
||||
server.launch().await;
|
||||
wait_until_gate_is_polled(&gate);
|
||||
std::fs::write(&gate, "go").expect("the gate opens");
|
||||
wait_for_success(&server, &run_id).await;
|
||||
|
||||
let (restored, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(restored, path);
|
||||
assert_eq!(subjects(&commits), three_stage_subjects(&run_id));
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(restored.join("one.txt")).expect("one.txt was restored"),
|
||||
"one\n"
|
||||
);
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A stage that fails on its own terms routes to its failure edge on the
|
||||
/// committed files, and after a crash once the failure is durable the
|
||||
/// route reruns on the same files.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_failure_route_sees_the_same_committed_files_after_a_crash() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let gate = context.temp_dir.join("fix.gate");
|
||||
let workspace = write_petri_workflow(
|
||||
&context,
|
||||
&format!(
|
||||
"digraph Failure {{\n graph [goal=\"Route on failure\", default_max_retries=0]\n \
|
||||
start [shape=Mdiamond]\n exit [shape=Msquare]\n work [shape=parallelogram, \
|
||||
script=\"echo partial > out.txt; exit 1\"]\n fix [shape=parallelogram, script=\"test \
|
||||
\\\"$(cat out.txt)\\\" = partial && echo run >> fix.log && {} && echo fixed >> \
|
||||
out.txt\"]\n start -> work -> exit\n work -> fix [condition=\"outcome=failed\"]\n \
|
||||
fix -> exit\n}}\n",
|
||||
wait_for(&gate)
|
||||
),
|
||||
);
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
|
||||
wait_for_status(&server, &run_id, &["running"]).await;
|
||||
let worker = wait_for_worker(&run_id);
|
||||
wait_until_gate_is_polled(&gate);
|
||||
// The route is running on the committed failure: the crash lands here.
|
||||
crash(&mut server, worker, Some(&gate));
|
||||
|
||||
server.launch().await;
|
||||
wait_until_gate_is_polled(&gate);
|
||||
std::fs::write(&gate, "go").expect("the gate opens");
|
||||
wait_for_success(&server, &run_id).await;
|
||||
|
||||
let (path, commits) = server.workspace_commits(&run_id);
|
||||
assert_eq!(subjects(&commits), vec![
|
||||
format!("fabro({run_id}): start (success)"),
|
||||
format!("fabro({run_id}): work (failure)"),
|
||||
format!("fabro({run_id}): fix (success)"),
|
||||
format!("fabro({run_id}): exit (success)"),
|
||||
]);
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(path.join("out.txt")).expect("out.txt"),
|
||||
"partial\nfixed\n"
|
||||
);
|
||||
assert_eq!(
|
||||
std::fs::read_to_string(path.join("fix.log")).expect("fix.log"),
|
||||
"run\n",
|
||||
"the route saw the failed stage's files, not its own interrupted attempt's"
|
||||
);
|
||||
server.shutdown();
|
||||
}
|
||||
|
||||
/// A checkpoint commit that fails ends the run: `checkpoint_failed` is
|
||||
/// recorded, no route runs, the run is reported failed, and a restart
|
||||
/// leaves it failed without launching a worker.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn a_failed_checkpoint_fails_the_run_and_a_restart_leaves_it_failed() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let context = test_context!();
|
||||
let mut server = RunningServer::start().await;
|
||||
let workspace = write_petri_workflow(
|
||||
&context,
|
||||
"digraph Wreck {\n graph [goal=\"Wreck the repository\", default_max_retries=0]\n start \
|
||||
[shape=Mdiamond]\n exit [shape=Msquare]\n wreck [shape=parallelogram, script=\"rm -rf \
|
||||
.git && echo garbage > .git\"]\n next [shape=parallelogram, script=\"echo next > \
|
||||
next.txt\"]\n fix [shape=parallelogram, script=\"echo fix > fix.txt\"]\n start -> wreck \
|
||||
-> next -> exit\n wreck -> fix [condition=\"outcome=failed\"]\n fix -> exit\n}\n",
|
||||
);
|
||||
let run_id = run_detached(&context, &server, &workspace);
|
||||
let status = wait_for_status(&server, &run_id, &["succeeded", "failed"]).await;
|
||||
let run = run_json(&server, &format!("runs/{run_id}")).await;
|
||||
assert_eq!(status, "failed", "run: {run}");
|
||||
let failures: Vec<String> = settled_events(&server, &run_id, "run.failed")
|
||||
.await
|
||||
.iter()
|
||||
.filter_map(|envelope| match &envelope.event.body {
|
||||
EventBody::RunFailed(props) => Some(props.failure.detail.message.clone()),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(failures.len(), 1, "{failures:?}");
|
||||
assert!(
|
||||
failures[0].contains("checkpoint commit of `wreck` failed"),
|
||||
"{failures:?}"
|
||||
);
|
||||
let scopes = server.petri_run_dir(&run_id).join("scopes");
|
||||
let work = std::fs::read_dir(&scopes)
|
||||
.expect("the scopes directory lists")
|
||||
.map(|entry| entry.expect("an entry reads").path().join("work"))
|
||||
.next()
|
||||
.expect("one workspace");
|
||||
assert!(!work.join("next.txt").exists(), "no route ran");
|
||||
assert!(!work.join("fix.txt").exists(), "no route ran");
|
||||
|
||||
let store = server.petri_store().await;
|
||||
let outcome = engine::outcome_of(&store, &run_id)
|
||||
.await
|
||||
.expect("the run's Petri record inspects");
|
||||
assert_ne!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
let checkpoints = server.checkpoints(&run_id).await;
|
||||
assert_eq!(
|
||||
checkpoints.len(),
|
||||
1,
|
||||
"only start was recorded: {checkpoints:?}"
|
||||
);
|
||||
|
||||
// The restart finds the run terminal and launches nothing for it.
|
||||
server.kill();
|
||||
server.launch().await;
|
||||
assert_eq!(run_status(&server, &run_id).await, "failed");
|
||||
std::thread::sleep(Duration::from_secs(1));
|
||||
assert_eq!(
|
||||
worker_pid(&run_id),
|
||||
None,
|
||||
"no worker was launched for the failed run"
|
||||
);
|
||||
server.shutdown();
|
||||
}
|
||||
|
|
|
|||
|
|
@ -18,11 +18,13 @@ use std::sync::Arc;
|
|||
use axum::extract::DefaultBodyLimit;
|
||||
use axum::routing::{get, post};
|
||||
use fabro_api::types::{
|
||||
PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriOpenResponse, PetriRecord,
|
||||
PetriRecordList, PetriReleaseRequest, WriteBlobResponse,
|
||||
PetriAccess, PetriAppendRequest, PetriOpenRequest, PetriOpenResponse, PetriPlatformRecord,
|
||||
PetriPlatformRecordAppendRequest, PetriPlatformRecordList, PetriRecord, PetriRecordList,
|
||||
PetriReleaseRequest, WriteBlobResponse,
|
||||
};
|
||||
use fabro_petri::petri::{Access, Digest, OwnerId, Record, StoreError};
|
||||
use fabro_petri::run_store::{log_id_text, parse_log_id};
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord};
|
||||
use fabro_types::BlobHash;
|
||||
use fabro_util::error::collect_chain;
|
||||
use serde_json::{Map, Value, json};
|
||||
|
|
@ -51,6 +53,10 @@ pub(super) fn routes() -> Router<Arc<AppState>> {
|
|||
post(write_blob).layer(DefaultBodyLimit::disable()),
|
||||
)
|
||||
.route("/runs/{id}/petri/blobs/{blobHash}", get(read_blob))
|
||||
.route(
|
||||
"/runs/{id}/petri/platform-records",
|
||||
get(list_platform_records).post(append_platform_record),
|
||||
)
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
|
|
@ -58,6 +64,11 @@ struct OwnerQuery {
|
|||
owner: String,
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct KindQuery {
|
||||
kind: Option<String>,
|
||||
}
|
||||
|
||||
async fn open_run(
|
||||
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
|
||||
State(state): State<Arc<AppState>>,
|
||||
|
|
@ -211,6 +222,115 @@ async fn read_blob(
|
|||
}
|
||||
}
|
||||
|
||||
/// The run's platform records, of one kind when the query names it.
|
||||
async fn list_platform_records(
|
||||
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
|
||||
State(state): State<Arc<AppState>>,
|
||||
Query(query): Query<KindQuery>,
|
||||
) -> Response {
|
||||
let store = state.stores.run_summaries.platform_records();
|
||||
let records = match query.kind.as_deref() {
|
||||
Some(kind) => match kind.parse::<PlatformRecordKind>() {
|
||||
Ok(kind) => store.read_kind(&id, kind).await,
|
||||
Err(_) => {
|
||||
return ApiError::bad_request(format!("`{kind}` is not a platform record kind."))
|
||||
.into_response();
|
||||
}
|
||||
},
|
||||
None => store.read(&id).await,
|
||||
};
|
||||
match records {
|
||||
Ok(records) => match records
|
||||
.into_iter()
|
||||
.map(|stored| wire_platform_record(&stored))
|
||||
.collect::<Result<Vec<_>, _>>()
|
||||
{
|
||||
Ok(records) => Json(PetriPlatformRecordList { records }).into_response(),
|
||||
Err(err) => err.into_response(),
|
||||
},
|
||||
Err(err) => platform_store_error_response(id, &err),
|
||||
}
|
||||
}
|
||||
|
||||
/// Store one platform record for the run and wake its projector.
|
||||
async fn append_platform_record(
|
||||
RequireWorkerRunScoped(id): RequireWorkerRunScoped,
|
||||
State(state): State<Arc<AppState>>,
|
||||
Json(request): Json<PetriPlatformRecordAppendRequest>,
|
||||
) -> Response {
|
||||
let record: PlatformRecord = match serde_json::from_value(Value::Object(request.record)) {
|
||||
Ok(record) => record,
|
||||
Err(err) => {
|
||||
return ApiError::bad_request(format!("Invalid platform record: {err}"))
|
||||
.into_response();
|
||||
}
|
||||
};
|
||||
let position = match (request.execution, request.firing) {
|
||||
(Some(execution), Some(firing)) => Some(StagePosition { execution, firing }),
|
||||
_ => None,
|
||||
};
|
||||
let summaries = &state.stores.run_summaries;
|
||||
match summaries
|
||||
.platform_records()
|
||||
.append(&id, &record, position)
|
||||
.await
|
||||
{
|
||||
Ok(stored) => {
|
||||
summaries.notify_platform_record(id);
|
||||
match wire_platform_record(&stored) {
|
||||
Ok(record) => Json(record).into_response(),
|
||||
Err(err) => err.into_response(),
|
||||
}
|
||||
}
|
||||
Err(err) => platform_store_error_response(id, &err),
|
||||
}
|
||||
}
|
||||
|
||||
/// A stored platform record as the wire carries it.
|
||||
fn wire_platform_record(stored: &StoredPlatformRecord) -> Result<PetriPlatformRecord, ApiError> {
|
||||
let record = serde_json::to_value(&stored.record).map_err(|err| {
|
||||
ApiError::with_code(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!(
|
||||
"The stored platform record at seq {} does not encode: {err}",
|
||||
stored.seq
|
||||
),
|
||||
"petri_store_failed",
|
||||
)
|
||||
})?;
|
||||
let Value::Object(record) = record else {
|
||||
return Err(ApiError::with_code(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!(
|
||||
"The stored platform record at seq {} is not a JSON object.",
|
||||
stored.seq
|
||||
),
|
||||
"petri_store_failed",
|
||||
));
|
||||
};
|
||||
Ok(PetriPlatformRecord {
|
||||
seq: stored.seq,
|
||||
recorded_at: stored.recorded_at,
|
||||
record,
|
||||
execution: stored.position.map(|position| position.execution),
|
||||
firing: stored.position.map(|position| position.firing),
|
||||
})
|
||||
}
|
||||
|
||||
fn platform_store_error_response(run_id: RunId, err: &fabro_store::Error) -> Response {
|
||||
tracing::error!(
|
||||
run_id = %run_id,
|
||||
error = %collect_chain(err).join(": "),
|
||||
"platform record store failed"
|
||||
);
|
||||
ApiError::with_code(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
"The platform record store failed; see the server log.",
|
||||
"petri_store_failed",
|
||||
)
|
||||
.into_response()
|
||||
}
|
||||
|
||||
fn unknown_log(log: &str) -> Response {
|
||||
ApiError::bad_request(format!(
|
||||
"`{log}` is not a Petri log: expected `coordinator`, `resources` or `execution <n>`."
|
||||
|
|
|
|||
|
|
@ -23,7 +23,10 @@
|
|||
//! item that follows.
|
||||
//!
|
||||
//! After a server restart, [`reconcile_on_startup`] hands a Petri run the
|
||||
//! previous server left in flight back to a worker in resume mode.
|
||||
//! previous server left in flight back to a worker in resume mode, once the
|
||||
//! recovery protocol (`fabro_petri::recovery`) has brought every live
|
||||
//! workspace to the snapshot its durable state names, or reports the run
|
||||
//! failed when it cannot.
|
||||
|
||||
use std::collections::{BTreeMap, HashSet};
|
||||
use std::sync::Arc;
|
||||
|
|
@ -34,8 +37,11 @@ use fabro_interview::ControlInterviewer;
|
|||
use fabro_llm::selection;
|
||||
use fabro_petri::check::{self, Bundle, CheckError, CheckRequest, Diagnostic, Launch};
|
||||
use fabro_petri::engine::{self, Conclusion, Execution, RunRequest};
|
||||
use fabro_petri::hooks::HooksSpec;
|
||||
use fabro_petri::interview::{Approval, DatabaseQuestions, FabroInterviewer};
|
||||
use fabro_petri::petri::StoreError;
|
||||
use fabro_petri::platform_records::SqlitePlatformRecords;
|
||||
use fabro_petri::recovery::{self, Recovery, RecoveryRequest};
|
||||
use fabro_petri::runtime::{self, RuntimeSpec};
|
||||
use fabro_petri::secrets::VaultSecrets;
|
||||
use fabro_petri::{SqliteRunStore, admission};
|
||||
|
|
@ -378,6 +384,12 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
|
|||
let observers = vec![petri_interviewer.observer()];
|
||||
let (_, eligible) = state.resolve_llm_client_with_ready_ids().await;
|
||||
let dry_run = run_state.spec.settings.run.execution.mode == RunMode::DryRun;
|
||||
let hooks = HooksSpec::for_run(
|
||||
Arc::new(SqlitePlatformRecords::new(Arc::clone(
|
||||
&state.stores.run_summaries,
|
||||
))),
|
||||
&run_state.spec.settings.run,
|
||||
);
|
||||
let request = RunRequest {
|
||||
run_id: run_id.to_string(),
|
||||
run_dir: run_dir.join("petri"),
|
||||
|
|
@ -392,6 +404,7 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
|
|||
observers,
|
||||
secrets: Some(Arc::new(VaultSecrets::from_vault(&vault))),
|
||||
blobs: Some(state.store_ref().blobs()),
|
||||
hooks: Some(hooks),
|
||||
};
|
||||
let result = Box::pin(engine::run(request)).await;
|
||||
let timing = RunTiming {
|
||||
|
|
@ -430,21 +443,22 @@ pub(crate) async fn execute(state: Arc<AppState>, run_id: RunId) {
|
|||
}
|
||||
|
||||
/// Bring a Petri run the server left in flight back to its worker after a
|
||||
/// restart: the run continues from its records, as Petri's own resume does.
|
||||
/// restart: the run continues from its records, as Petri's own resume does,
|
||||
/// on workspaces that match them.
|
||||
///
|
||||
/// The lease the previous worker held is released from outside, which
|
||||
/// fences that worker should it still be alive; then the run is asked to
|
||||
/// start again as a resume (`run.start_requested` with `resume`, then
|
||||
/// `run.runnable`, the same pair the API's resume appends), and a managed
|
||||
/// run is registered for the scheduler in resume mode when Petri's store
|
||||
/// holds the run, else in start mode: a worker that died before it created
|
||||
/// the run's record left nothing to continue from, so the run starts from
|
||||
/// its admitted graphs.
|
||||
///
|
||||
/// Full recovery, where the workspace a resumed stage sees is restored to
|
||||
/// the snapshot its durable state names, is the integration plan's F3.5.
|
||||
/// Until it lands, a retained workspace is used as the previous worker left
|
||||
/// it.
|
||||
/// fences that worker should it still be alive. Then the recovery protocol
|
||||
/// reads the run's durable execution state: a run with a failed checkpoint
|
||||
/// is reported failed here and never resumed; otherwise every live
|
||||
/// workspace on this host is verified against, reset to, or restored from
|
||||
/// the snapshot its last durable finish names, and a finish with no
|
||||
/// snapshot fails the run rather than resume it on stale files. The run is
|
||||
/// then asked to start again as a resume (`run.start_requested` with
|
||||
/// `resume`, then `run.runnable`, the same pair the API's resume appends),
|
||||
/// and a managed run is registered for the scheduler in resume mode when
|
||||
/// Petri's store holds the run, else in start mode: a worker that died
|
||||
/// before it created the run's record left nothing to continue from, so the
|
||||
/// run starts from its admitted graphs.
|
||||
pub(crate) async fn reconcile_on_startup(
|
||||
state: &Arc<AppState>,
|
||||
run_id: RunId,
|
||||
|
|
@ -459,8 +473,46 @@ pub(crate) async fn reconcile_on_startup(
|
|||
return Err(anyhow::Error::new(err).context("releasing the Petri run's lease"));
|
||||
}
|
||||
};
|
||||
let run_dir = Storage::new(state.server_storage_dir())
|
||||
.run_scratch(&run_id)
|
||||
.root()
|
||||
.to_path_buf();
|
||||
let mode = if held {
|
||||
RunExecutionMode::Resume
|
||||
let request = RecoveryRequest::for_run(
|
||||
run_id,
|
||||
run_dir.join("petri"),
|
||||
Arc::new(SqliteRunStore::new(state.db_pool.clone())),
|
||||
Arc::new(SqlitePlatformRecords::new(Arc::clone(
|
||||
&state.stores.run_summaries,
|
||||
))),
|
||||
&run_state.spec.settings.run,
|
||||
);
|
||||
match recovery::recover(request)
|
||||
.await
|
||||
.map_err(|err| anyhow::Error::new(err).context("recovering the Petri run"))?
|
||||
{
|
||||
Recovery::Start => RunExecutionMode::Start,
|
||||
Recovery::Resume { workspaces } => {
|
||||
info!(
|
||||
run_id = %run_id,
|
||||
workspaces = workspaces.len(),
|
||||
"Petri run's workspaces match its durable state"
|
||||
);
|
||||
RunExecutionMode::Resume
|
||||
}
|
||||
Recovery::Failed { reason } => {
|
||||
warn!(
|
||||
run_id = %run_id,
|
||||
petri_key = %key,
|
||||
error = %reason,
|
||||
"Petri run left in flight by the previous server cannot resume; reporting it failed"
|
||||
);
|
||||
let (_, _, event) =
|
||||
failed(FailureReason::WorkflowError, reason, RunTiming::default());
|
||||
workflow_event::append_event(run_store, &run_id, &event).await?;
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
} else {
|
||||
RunExecutionMode::Start
|
||||
};
|
||||
|
|
@ -482,10 +534,6 @@ pub(crate) async fn reconcile_on_startup(
|
|||
] {
|
||||
workflow_event::append_event(run_store, &run_id, &event).await?;
|
||||
}
|
||||
let run_dir = Storage::new(state.server_storage_dir())
|
||||
.run_scratch(&run_id)
|
||||
.root()
|
||||
.to_path_buf();
|
||||
let mut runs = state.runs.lock().expect("runs lock poisoned");
|
||||
runs.insert(
|
||||
run_id,
|
||||
|
|
|
|||
|
|
@ -62,6 +62,9 @@ const WORKER_ENV_ALLOWLIST: &[&str] = &[
|
|||
EnvVars::PETRI_SANDBOX_PLUGIN_DEV,
|
||||
EnvVars::PETRI_SANDBOX_DOCKER_HOST_ADDRESS,
|
||||
EnvVars::PETRI_SANDBOX_ACTION_HOST_IMAGE,
|
||||
// A test's checkpoint gates: the worker's hooks hold at a named point
|
||||
// until the test releases them, so a crash can be placed there.
|
||||
EnvVars::FABRO_TEST_CHECKPOINT_GATES,
|
||||
];
|
||||
|
||||
const RENDER_GRAPH_ENV_ALLOWLIST: &[&str] = &[EnvVars::PATH, EnvVars::HOME, EnvVars::TMPDIR];
|
||||
|
|
|
|||
|
|
@ -346,3 +346,91 @@ async fn a_worker_store_leases_for_its_launch_not_for_petris_owner() {
|
|||
drop(created);
|
||||
wait_until_released(server_store, &key).await;
|
||||
}
|
||||
|
||||
/// A worker stores Fabro's platform records for its run over the API and
|
||||
/// reads them back by kind: what the checkpoint hooks do from the worker
|
||||
/// process.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn a_worker_appends_and_reads_platform_records_over_the_api() {
|
||||
use fabro_petri::platform_records::{HttpPlatformRecords, PlatformRecords};
|
||||
use fabro_store::platform_records::{CheckpointRecord, DecisionRef, OperationKey};
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
|
||||
|
||||
let state = test_app_state();
|
||||
let base_url = serve(Arc::clone(&state), |router| router).await;
|
||||
let run_id = RunId::new();
|
||||
let token = state.test_issue_worker_token(&run_id);
|
||||
let records =
|
||||
HttpPlatformRecords::new(worker_client(&base_url, &token, Duration::from_secs(5)).await);
|
||||
|
||||
let checkpoint = PlatformRecord::Checkpoint(CheckpointRecord {
|
||||
execution: 0,
|
||||
firing: 3,
|
||||
attempt: Some(1),
|
||||
workspace: Some("invocation-0-scope-0".to_string()),
|
||||
git_commit_sha: Some("abc123".to_string()),
|
||||
diff_summary: None,
|
||||
patch_blob: None,
|
||||
operation: Some(OperationKey {
|
||||
execution: 0,
|
||||
decision: DecisionRef::AttemptStart {
|
||||
firing: 3,
|
||||
attempt: 1,
|
||||
},
|
||||
effect: "checkpoint".to_string(),
|
||||
}),
|
||||
});
|
||||
let position = Some(StagePosition {
|
||||
execution: 0,
|
||||
firing: 3,
|
||||
});
|
||||
let stored = records
|
||||
.append(&run_id, &checkpoint, position)
|
||||
.await
|
||||
.expect("the record appends over the API");
|
||||
assert_eq!(stored.seq, 1);
|
||||
assert_eq!(stored.position, position);
|
||||
let notice = PlatformRecord::RunArchived;
|
||||
records
|
||||
.append(&run_id, ¬ice, None)
|
||||
.await
|
||||
.expect("a second record appends");
|
||||
|
||||
let checkpoints = records
|
||||
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.expect("the checkpoints read back");
|
||||
assert_eq!(checkpoints.len(), 1);
|
||||
assert_eq!(checkpoints[0].seq, 1);
|
||||
assert_eq!(checkpoints[0].position, position);
|
||||
assert!(matches!(
|
||||
&checkpoints[0].record,
|
||||
PlatformRecord::Checkpoint(record) if record.git_commit_sha.as_deref() == Some("abc123")
|
||||
&& record.operation == checkpoint_operation(&checkpoint)
|
||||
));
|
||||
let archived = records
|
||||
.read_kind(&run_id, PlatformRecordKind::RunArchived)
|
||||
.await
|
||||
.expect("the archive record reads back");
|
||||
assert_eq!(archived.len(), 1);
|
||||
assert_eq!(archived[0].seq, 2);
|
||||
assert_eq!(archived[0].position, None);
|
||||
|
||||
// Another run's token cannot read this run's records.
|
||||
let other = state.test_issue_worker_token(&RunId::new());
|
||||
let foreign =
|
||||
HttpPlatformRecords::new(worker_client(&base_url, &other, Duration::from_secs(5)).await);
|
||||
assert!(
|
||||
foreign
|
||||
.read_kind(&run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.is_err(),
|
||||
"a worker token names one run"
|
||||
);
|
||||
}
|
||||
|
||||
fn checkpoint_operation(
|
||||
record: &fabro_store::PlatformRecord,
|
||||
) -> Option<fabro_store::platform_records::OperationKey> {
|
||||
record.operation().cloned()
|
||||
}
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ fabro-store = { path = "../fabro-store" }
|
|||
fabro-types = { path = "../../foundation/fabro-types" }
|
||||
fabro-vault = { path = "../../foundation/fabro-vault" }
|
||||
fabro-workflow = { path = "../fabro-workflow" }
|
||||
fabro-checkpoint = { path = "../fabro-checkpoint" }
|
||||
fabro-util = { path = "../../foundation/fabro-util" }
|
||||
petri_runtime.workspace = true
|
||||
petri_execution.workspace = true
|
||||
|
|
@ -50,6 +51,7 @@ tokio-util.workspace = true
|
|||
tracing.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
fabro-petri = { path = ".", features = ["test-support"] }
|
||||
fabro-auth = { path = "../../foundation/fabro-auth", features = ["test-support"] }
|
||||
fabro-llm = { path = "../fabro-llm", features = ["test-support"] }
|
||||
fabro-store = { path = "../fabro-store", features = ["test-support"] }
|
||||
|
|
@ -57,4 +59,5 @@ fabro-test.workspace = true
|
|||
fabro-types = { path = "../../foundation/fabro-types", features = ["test-support"] }
|
||||
petri_testkit.workspace = true
|
||||
tempfile = "3"
|
||||
lithos-llm = { workspace = true, features = ["runtime"] }
|
||||
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }
|
||||
|
|
|
|||
864
lib/components/fabro-petri/src/checkpoint.rs
Normal file
864
lib/components/fabro-petri/src/checkpoint.rs
Normal file
|
|
@ -0,0 +1,864 @@
|
|||
//! Git snapshots of a Petri run's workspaces: the checkpoint commit and
|
||||
//! what recovery does with it.
|
||||
//!
|
||||
//! A Fabro stage's files are committed on the run branch of its workspace
|
||||
//! before the stage's finish is recorded, so a durable finish implies a
|
||||
//! durable snapshot (the integration plan's F3.1). The commit message
|
||||
//! carries the snapshot's identity as trailers, the run key, execution,
|
||||
//! firing and attempt, so a restart reconciles a missing platform record
|
||||
//! from the branch alone.
|
||||
//!
|
||||
//! # Where the workspace is
|
||||
//!
|
||||
//! Petri's host backend keeps a scope's workspace under the run directory
|
||||
//! at `scopes/<workspace id>/work`, the layout `HostExecutor::workspace_for`
|
||||
//! names. This module reaches it there and runs `git` on the host, which
|
||||
//! is where the worker, and the server at recovery, run. A Docker or
|
||||
//! Daytona workspace lives inside its sandbox, out of reach of this module:
|
||||
//! the hooks record that no snapshot was taken and recovery resumes such a
|
||||
//! run on the retained sandbox as it was left.
|
||||
//!
|
||||
//! # The snapshot repository
|
||||
//!
|
||||
//! Every checkpoint commit is also pushed to a bare repository beside the
|
||||
//! run's workspaces, `snapshots/<workspace id>.git`, under an immutable ref
|
||||
//! per checkpoint (`refs/checkpoints/<execution>/<firing>/<attempt>`). A
|
||||
//! workspace that is gone at recovery is restored from it, and the refs
|
||||
//! are what recovery reconciles a missing record from.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Stdio;
|
||||
use std::time::Duration;
|
||||
|
||||
use fabro_checkpoint::author::GitAuthor;
|
||||
use fabro_checkpoint::trailer::{self, Trailer};
|
||||
use fabro_store::platform_records::{DecisionRef, OperationKey};
|
||||
use fabro_types::settings::run::RunCheckpointSettings;
|
||||
use tokio::process::Command;
|
||||
use tokio::{fs, time};
|
||||
|
||||
/// The failure class of a stage whose checkpoint commit failed: fatal to
|
||||
/// the run, and terminal for a restart.
|
||||
pub const CHECKPOINT_FAILED_CLASS: &str = "checkpoint_failed";
|
||||
|
||||
/// The effect kind of a checkpoint in its operation identity.
|
||||
pub const CHECKPOINT_EFFECT: &str = "checkpoint";
|
||||
|
||||
pub const RUN_TRAILER: &str = "Fabro-Run";
|
||||
pub const EXECUTION_TRAILER: &str = "Fabro-Execution";
|
||||
pub const FIRING_TRAILER: &str = "Fabro-Firing";
|
||||
pub const ATTEMPT_TRAILER: &str = "Fabro-Attempt";
|
||||
|
||||
const FOOTER: &str = "\u{2692}\u{fe0f} Generated with [Fabro](https://fabro.sh)";
|
||||
const REFS_PREFIX: &str = "refs/checkpoints/";
|
||||
|
||||
/// Directories never committed, the legacy executor's list: build output
|
||||
/// and dependency caches a stage regenerates.
|
||||
pub const EXCLUDE_DIRS: &[&str] = &[
|
||||
".git",
|
||||
"node_modules",
|
||||
".pnpm-store",
|
||||
".npm",
|
||||
"target",
|
||||
".next",
|
||||
"__pycache__",
|
||||
".venv",
|
||||
"venv",
|
||||
".cache",
|
||||
".tox",
|
||||
".pytest_cache",
|
||||
];
|
||||
|
||||
/// The identity of one snapshot: the attempt whose files it holds.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)]
|
||||
pub struct CheckpointKey {
|
||||
pub execution: u64,
|
||||
pub firing: u64,
|
||||
pub attempt: u32,
|
||||
}
|
||||
|
||||
impl CheckpointKey {
|
||||
/// The immutable ref the snapshot is published under.
|
||||
#[must_use]
|
||||
pub fn snapshot_ref(self) -> String {
|
||||
format!(
|
||||
"{REFS_PREFIX}{}/{}/{}",
|
||||
self.execution, self.firing, self.attempt
|
||||
)
|
||||
}
|
||||
|
||||
/// The operation identity of the checkpoint effect: the attempt's
|
||||
/// decision in its execution, effect kind `checkpoint`.
|
||||
#[must_use]
|
||||
pub fn operation(self) -> OperationKey {
|
||||
OperationKey {
|
||||
execution: self.execution,
|
||||
decision: DecisionRef::AttemptStart {
|
||||
firing: self.firing,
|
||||
attempt: self.attempt,
|
||||
},
|
||||
effect: CHECKPOINT_EFFECT.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The key an operation identity names, when it is a checkpoint's.
|
||||
#[must_use]
|
||||
pub fn from_operation(operation: &OperationKey) -> Option<Self> {
|
||||
match operation.decision {
|
||||
DecisionRef::AttemptStart { firing, attempt }
|
||||
if operation.effect == CHECKPOINT_EFFECT =>
|
||||
{
|
||||
Some(Self {
|
||||
execution: operation.execution,
|
||||
firing,
|
||||
attempt,
|
||||
})
|
||||
}
|
||||
DecisionRef::AttemptStart { .. }
|
||||
| DecisionRef::ExecutionStart
|
||||
| DecisionRef::Route { .. } => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn from_ref(name: &str) -> Option<Self> {
|
||||
let mut parts = name.strip_prefix(REFS_PREFIX)?.split('/');
|
||||
let execution = parts.next()?.parse().ok()?;
|
||||
let firing = parts.next()?.parse().ok()?;
|
||||
let attempt = parts.next()?.parse().ok()?;
|
||||
parts.next().is_none().then_some(Self {
|
||||
execution,
|
||||
firing,
|
||||
attempt,
|
||||
})
|
||||
}
|
||||
|
||||
/// The key a checkpoint commit's message carries in its trailers.
|
||||
#[must_use]
|
||||
pub fn from_message(message: &str) -> Option<Self> {
|
||||
Some(Self {
|
||||
execution: trailer::parse(message, EXECUTION_TRAILER)?.parse().ok()?,
|
||||
firing: trailer::parse(message, FIRING_TRAILER)?.parse().ok()?,
|
||||
attempt: trailer::parse(message, ATTEMPT_TRAILER)?.parse().ok()?,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a snapshot could not be taken, found or restored.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum CheckpointError {
|
||||
#[error("the workspace `{workspace}` does not exist at {}", path.display())]
|
||||
WorkspaceMissing {
|
||||
workspace: String,
|
||||
path: PathBuf,
|
||||
},
|
||||
#[error("git {action} failed ({status}): {detail}")]
|
||||
Command {
|
||||
action: String,
|
||||
status: String,
|
||||
detail: String,
|
||||
},
|
||||
#[error("git {action} could not run")]
|
||||
Spawn {
|
||||
action: String,
|
||||
#[source]
|
||||
source: std::io::Error,
|
||||
},
|
||||
#[error("git {action} did not finish within {timeout:?}")]
|
||||
TimedOut { action: String, timeout: Duration },
|
||||
#[error("the workspace could not be prepared at {}", path.display())]
|
||||
Io {
|
||||
path: PathBuf,
|
||||
#[source]
|
||||
source: std::io::Error,
|
||||
},
|
||||
#[error("the restored workspace is at {actual}, not the snapshot {expected}")]
|
||||
RestoreMismatch { expected: String, actual: String },
|
||||
}
|
||||
|
||||
/// A checkpoint commit: the commit, and whether an earlier attempt of the
|
||||
/// same operation had already made it.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct Snapshot {
|
||||
pub sha: String,
|
||||
pub reused: bool,
|
||||
}
|
||||
|
||||
/// One published snapshot of a workspace.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct PublishedSnapshot {
|
||||
pub key: CheckpointKey,
|
||||
pub sha: String,
|
||||
}
|
||||
|
||||
/// The workspaces of one run on this host, and the Git operations Fabro
|
||||
/// performs on them.
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RunWorkspaces {
|
||||
run_dir: PathBuf,
|
||||
run_id: String,
|
||||
author: GitAuthor,
|
||||
exclude_globs: Vec<String>,
|
||||
timeout: Duration,
|
||||
}
|
||||
|
||||
impl RunWorkspaces {
|
||||
#[must_use]
|
||||
pub fn new(
|
||||
run_dir: PathBuf,
|
||||
run_id: String,
|
||||
author: GitAuthor,
|
||||
settings: &RunCheckpointSettings,
|
||||
) -> Self {
|
||||
Self {
|
||||
run_dir,
|
||||
run_id,
|
||||
author,
|
||||
exclude_globs: settings.exclude_globs.clone(),
|
||||
timeout: Duration::from_millis(settings.commit_timeout_ms.max(1)),
|
||||
}
|
||||
}
|
||||
|
||||
/// The run branch every workspace of the run commits on.
|
||||
#[must_use]
|
||||
pub fn run_branch(&self) -> String {
|
||||
format!("fabro/run/{}", self.run_id)
|
||||
}
|
||||
|
||||
/// Where the host backend keeps the workspace: `scopes/<id>/work` under
|
||||
/// the run directory.
|
||||
#[must_use]
|
||||
pub fn workspace_path(&self, workspace: &str) -> PathBuf {
|
||||
self.run_dir.join("scopes").join(workspace).join("work")
|
||||
}
|
||||
|
||||
/// The bare repository the workspace's snapshots are published to.
|
||||
#[must_use]
|
||||
pub fn snapshot_repository(&self, workspace: &str) -> PathBuf {
|
||||
self.run_dir
|
||||
.join("snapshots")
|
||||
.join(format!("{workspace}.git"))
|
||||
}
|
||||
|
||||
/// Whether the workspace exists on this host.
|
||||
pub async fn workspace_exists(&self, workspace: &str) -> bool {
|
||||
fs::try_exists(self.workspace_path(workspace))
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Commit the workspace's files on the run branch as the snapshot of
|
||||
/// `key`, and publish it. An earlier commit of the same key that the
|
||||
/// workspace still sits on, unchanged, is reused.
|
||||
pub async fn commit(
|
||||
&self,
|
||||
workspace: &str,
|
||||
key: CheckpointKey,
|
||||
node: &str,
|
||||
status: &str,
|
||||
) -> Result<Snapshot, CheckpointError> {
|
||||
let path = self.workspace_path(workspace);
|
||||
if !self.workspace_exists(workspace).await {
|
||||
return Err(CheckpointError::WorkspaceMissing {
|
||||
workspace: workspace.to_string(),
|
||||
path,
|
||||
});
|
||||
}
|
||||
self.ensure_repository(&path).await?;
|
||||
if let Some(existing) = self.published_sha(workspace, key).await? {
|
||||
if self.head(&path).await?.as_deref() == Some(existing.as_str())
|
||||
&& self.is_clean(&path).await?
|
||||
{
|
||||
return Ok(Snapshot {
|
||||
sha: existing,
|
||||
reused: true,
|
||||
});
|
||||
}
|
||||
}
|
||||
let mut add = vec![
|
||||
"add".to_string(),
|
||||
"-A".to_string(),
|
||||
"--".to_string(),
|
||||
".".to_string(),
|
||||
];
|
||||
add.extend(
|
||||
EXCLUDE_DIRS
|
||||
.iter()
|
||||
.map(|dir| format!(":(glob,exclude)**/{dir}/**")),
|
||||
);
|
||||
add.extend(
|
||||
self.exclude_globs
|
||||
.iter()
|
||||
.map(|glob| format!(":(glob,exclude){glob}")),
|
||||
);
|
||||
self.git(&path, "add", &add).await?;
|
||||
let message = self.message(key, node, status);
|
||||
let user_name = format!("user.name={}", self.author.name);
|
||||
let user_email = format!("user.email={}", self.author.email);
|
||||
self.git(&path, "commit", &[
|
||||
"-c",
|
||||
&user_name,
|
||||
"-c",
|
||||
&user_email,
|
||||
"commit",
|
||||
"-q",
|
||||
"--allow-empty",
|
||||
"-m",
|
||||
&message,
|
||||
])
|
||||
.await?;
|
||||
let sha = self.git(&path, "rev-parse", &["rev-parse", "HEAD"]).await?;
|
||||
self.publish(workspace, &path, key, &sha).await?;
|
||||
Ok(Snapshot { sha, reused: false })
|
||||
}
|
||||
|
||||
/// The commit of `key`, from the snapshot repository first, else from
|
||||
/// the workspace's own history by the trailers.
|
||||
pub async fn find(
|
||||
&self,
|
||||
workspace: &str,
|
||||
key: CheckpointKey,
|
||||
) -> Result<Option<String>, CheckpointError> {
|
||||
if let Some(sha) = self.published_sha(workspace, key).await? {
|
||||
return Ok(Some(sha));
|
||||
}
|
||||
let path = self.workspace_path(workspace);
|
||||
if !self.workspace_exists(workspace).await || self.head(&path).await?.is_none() {
|
||||
return Ok(None);
|
||||
}
|
||||
let listed = self
|
||||
.git(&path, "log", &[
|
||||
"log",
|
||||
"--format=%H",
|
||||
"--extended-regexp",
|
||||
&format!("--grep=^{EXECUTION_TRAILER}: {}$", key.execution),
|
||||
&format!("--grep=^{FIRING_TRAILER}: {}$", key.firing),
|
||||
&format!("--grep=^{ATTEMPT_TRAILER}: {}$", key.attempt),
|
||||
"--all-match",
|
||||
"HEAD",
|
||||
])
|
||||
.await?;
|
||||
Ok(listed.lines().next().map(str::to_owned))
|
||||
}
|
||||
|
||||
/// Every snapshot published for the workspace.
|
||||
pub async fn published(
|
||||
&self,
|
||||
workspace: &str,
|
||||
) -> Result<Vec<PublishedSnapshot>, CheckpointError> {
|
||||
let repository = self.snapshot_repository(workspace);
|
||||
if !fs::try_exists(&repository).await.unwrap_or(false) {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let listed = self
|
||||
.git(&repository, "for-each-ref", &[
|
||||
"for-each-ref",
|
||||
"--format=%(refname) %(objectname)",
|
||||
REFS_PREFIX,
|
||||
])
|
||||
.await?;
|
||||
Ok(listed
|
||||
.lines()
|
||||
.filter_map(|line| {
|
||||
let (name, sha) = line.split_once(' ')?;
|
||||
Some(PublishedSnapshot {
|
||||
key: CheckpointKey::from_ref(name)?,
|
||||
sha: sha.to_string(),
|
||||
})
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Whether `ancestor` is reachable from `descendant` in the workspace's
|
||||
/// published history.
|
||||
pub async fn is_ancestor(
|
||||
&self,
|
||||
workspace: &str,
|
||||
ancestor: &str,
|
||||
descendant: &str,
|
||||
) -> Result<bool, CheckpointError> {
|
||||
let repository = self.snapshot_repository(workspace);
|
||||
Ok(self
|
||||
.git_status(&repository, "merge-base", &[
|
||||
"merge-base",
|
||||
"--is-ancestor",
|
||||
ancestor,
|
||||
descendant,
|
||||
])
|
||||
.await?
|
||||
.is_some())
|
||||
}
|
||||
|
||||
/// The workspace's `HEAD`, or `None` when it has no commit.
|
||||
pub async fn workspace_head(&self, workspace: &str) -> Result<Option<String>, CheckpointError> {
|
||||
let path = self.workspace_path(workspace);
|
||||
self.head(&path).await
|
||||
}
|
||||
|
||||
/// Whether the workspace sits on `sha` with nothing changed since.
|
||||
pub async fn matches(&self, workspace: &str, sha: &str) -> Result<bool, CheckpointError> {
|
||||
let path = self.workspace_path(workspace);
|
||||
Ok(self.head(&path).await?.as_deref() == Some(sha) && self.is_clean(&path).await?)
|
||||
}
|
||||
|
||||
/// Bring the workspace back to `sha`: tracked files reset, untracked
|
||||
/// files removed, the excluded caches left alone.
|
||||
pub async fn reset(&self, workspace: &str, sha: &str) -> Result<(), CheckpointError> {
|
||||
let path = self.workspace_path(workspace);
|
||||
self.git(&path, "reset", &["reset", "-q", "--hard", sha])
|
||||
.await?;
|
||||
let mut clean = vec!["clean".to_string(), "-fdq".to_string()];
|
||||
for dir in EXCLUDE_DIRS {
|
||||
clean.push("-e".to_string());
|
||||
clean.push((*dir).to_string());
|
||||
}
|
||||
for glob in &self.exclude_globs {
|
||||
clean.push("-e".to_string());
|
||||
clean.push(glob.clone());
|
||||
}
|
||||
self.git(&path, "clean", &clean).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Recreate a gone workspace from the published snapshot `key`, at
|
||||
/// `sha`, on the run branch.
|
||||
pub async fn restore(
|
||||
&self,
|
||||
workspace: &str,
|
||||
key: CheckpointKey,
|
||||
sha: &str,
|
||||
) -> Result<(), CheckpointError> {
|
||||
let path = self.workspace_path(workspace);
|
||||
fs::create_dir_all(&path)
|
||||
.await
|
||||
.map_err(|source| CheckpointError::Io {
|
||||
path: path.clone(),
|
||||
source,
|
||||
})?;
|
||||
self.git(&path, "init", &["init", "-q"]).await?;
|
||||
let repository = self.snapshot_repository(workspace);
|
||||
let repository = repository.to_string_lossy().into_owned();
|
||||
self.git(&path, "fetch", &[
|
||||
"fetch",
|
||||
"-q",
|
||||
&repository,
|
||||
&key.snapshot_ref(),
|
||||
])
|
||||
.await?;
|
||||
let branch = self.run_branch();
|
||||
self.git(&path, "checkout", &[
|
||||
"checkout",
|
||||
"-q",
|
||||
"-B",
|
||||
&branch,
|
||||
"FETCH_HEAD",
|
||||
])
|
||||
.await?;
|
||||
let actual = self.git(&path, "rev-parse", &["rev-parse", "HEAD"]).await?;
|
||||
if actual != sha {
|
||||
return Err(CheckpointError::RestoreMismatch {
|
||||
expected: sha.to_string(),
|
||||
actual,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The commit message: Fabro's subject, the footer, and the identity
|
||||
/// trailers last, so `git interpret-trailers` and
|
||||
/// [`CheckpointKey::from_message`] both read them.
|
||||
fn message(&self, key: CheckpointKey, node: &str, status: &str) -> String {
|
||||
let subject = format!("fabro({}): {node} ({status})", self.run_id);
|
||||
let execution = key.execution.to_string();
|
||||
let firing = key.firing.to_string();
|
||||
let attempt = key.attempt.to_string();
|
||||
let mut trailers = vec![
|
||||
Trailer {
|
||||
key: RUN_TRAILER,
|
||||
value: &self.run_id,
|
||||
},
|
||||
Trailer {
|
||||
key: EXECUTION_TRAILER,
|
||||
value: &execution,
|
||||
},
|
||||
Trailer {
|
||||
key: FIRING_TRAILER,
|
||||
value: &firing,
|
||||
},
|
||||
Trailer {
|
||||
key: ATTEMPT_TRAILER,
|
||||
value: &attempt,
|
||||
},
|
||||
];
|
||||
let defaults = GitAuthor::default();
|
||||
let co_author = format!("{} <{}>", defaults.name, defaults.email);
|
||||
if !self.author.is_default() {
|
||||
trailers.push(Trailer {
|
||||
key: "Co-Authored-By",
|
||||
value: &co_author,
|
||||
});
|
||||
}
|
||||
trailer::format_message(&subject, FOOTER, &trailers)
|
||||
}
|
||||
|
||||
/// A repository on the run branch, initialised when the workspace has
|
||||
/// none.
|
||||
async fn ensure_repository(&self, path: &Path) -> Result<(), CheckpointError> {
|
||||
if self
|
||||
.git_status(path, "rev-parse", &["rev-parse", "--git-dir"])
|
||||
.await?
|
||||
.is_none()
|
||||
{
|
||||
self.git(path, "init", &["init", "-q"]).await?;
|
||||
}
|
||||
let branch = self.run_branch();
|
||||
let current = self
|
||||
.git_status(path, "symbolic-ref", &[
|
||||
"symbolic-ref",
|
||||
"-q",
|
||||
"--short",
|
||||
"HEAD",
|
||||
])
|
||||
.await?;
|
||||
if current.as_deref() != Some(branch.as_str()) {
|
||||
self.git(path, "checkout", &["checkout", "-q", "-B", &branch])
|
||||
.await?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn publish(
|
||||
&self,
|
||||
workspace: &str,
|
||||
path: &Path,
|
||||
key: CheckpointKey,
|
||||
sha: &str,
|
||||
) -> Result<(), CheckpointError> {
|
||||
let repository = self.snapshot_repository(workspace);
|
||||
if !fs::try_exists(&repository).await.unwrap_or(false) {
|
||||
fs::create_dir_all(&repository)
|
||||
.await
|
||||
.map_err(|source| CheckpointError::Io {
|
||||
path: repository.clone(),
|
||||
source,
|
||||
})?;
|
||||
self.git(&repository, "init --bare", &["init", "-q", "--bare"])
|
||||
.await?;
|
||||
}
|
||||
let refspec = format!("{sha}:{}", key.snapshot_ref());
|
||||
let repository = repository.to_string_lossy().into_owned();
|
||||
self.git(path, "push", &[
|
||||
"push",
|
||||
"-q",
|
||||
"--force",
|
||||
&repository,
|
||||
&refspec,
|
||||
])
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn published_sha(
|
||||
&self,
|
||||
workspace: &str,
|
||||
key: CheckpointKey,
|
||||
) -> Result<Option<String>, CheckpointError> {
|
||||
let repository = self.snapshot_repository(workspace);
|
||||
if !fs::try_exists(&repository).await.unwrap_or(false) {
|
||||
return Ok(None);
|
||||
}
|
||||
self.git_status(&repository, "rev-parse", &[
|
||||
"rev-parse",
|
||||
"-q",
|
||||
"--verify",
|
||||
&key.snapshot_ref(),
|
||||
])
|
||||
.await
|
||||
}
|
||||
|
||||
async fn head(&self, path: &Path) -> Result<Option<String>, CheckpointError> {
|
||||
self.git_status(path, "rev-parse", &["rev-parse", "-q", "--verify", "HEAD"])
|
||||
.await
|
||||
}
|
||||
|
||||
async fn is_clean(&self, path: &Path) -> Result<bool, CheckpointError> {
|
||||
let status = self.git(path, "status", &["status", "--porcelain"]).await?;
|
||||
Ok(status.trim().is_empty())
|
||||
}
|
||||
|
||||
/// Run `git` in `cwd`; a non-zero exit is the error.
|
||||
async fn git<S: AsRef<str>>(
|
||||
&self,
|
||||
cwd: &Path,
|
||||
action: &str,
|
||||
args: &[S],
|
||||
) -> Result<String, CheckpointError> {
|
||||
let output = self.run(cwd, action, args).await?;
|
||||
if output.status.success() {
|
||||
Ok(String::from_utf8_lossy(&output.stdout).trim().to_owned())
|
||||
} else {
|
||||
Err(CheckpointError::Command {
|
||||
action: action.to_string(),
|
||||
status: output.status.to_string(),
|
||||
detail: detail(&output.stderr),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Run `git` in `cwd`; a non-zero exit is `None`, for the queries whose
|
||||
/// answer it is (an unborn `HEAD`, a missing ref, no repository).
|
||||
async fn git_status<S: AsRef<str>>(
|
||||
&self,
|
||||
cwd: &Path,
|
||||
action: &str,
|
||||
args: &[S],
|
||||
) -> Result<Option<String>, CheckpointError> {
|
||||
let output = self.run(cwd, action, args).await?;
|
||||
Ok(output
|
||||
.status
|
||||
.success()
|
||||
.then(|| String::from_utf8_lossy(&output.stdout).trim().to_owned()))
|
||||
}
|
||||
|
||||
async fn run<S: AsRef<str>>(
|
||||
&self,
|
||||
cwd: &Path,
|
||||
action: &str,
|
||||
args: &[S],
|
||||
) -> Result<std::process::Output, CheckpointError> {
|
||||
let mut command = Command::new("git");
|
||||
command
|
||||
.args([
|
||||
"-c",
|
||||
"core.hooksPath=/dev/null",
|
||||
"-c",
|
||||
"commit.gpgsign=false",
|
||||
"-c",
|
||||
"gc.auto=0",
|
||||
"-c",
|
||||
"advice.detachedHead=false",
|
||||
"-c",
|
||||
"init.defaultBranch=main",
|
||||
])
|
||||
.args(args.iter().map(AsRef::as_ref))
|
||||
.current_dir(cwd)
|
||||
.env("GIT_TERMINAL_PROMPT", "0")
|
||||
.env_remove("GIT_DIR")
|
||||
.env_remove("GIT_WORK_TREE")
|
||||
.env_remove("GIT_INDEX_FILE")
|
||||
.stdin(Stdio::null())
|
||||
.stdout(Stdio::piped())
|
||||
.stderr(Stdio::piped())
|
||||
.kill_on_drop(true);
|
||||
match time::timeout(self.timeout, command.output()).await {
|
||||
Ok(Ok(output)) => Ok(output),
|
||||
Ok(Err(source)) => Err(CheckpointError::Spawn {
|
||||
action: action.to_string(),
|
||||
source,
|
||||
}),
|
||||
Err(_) => Err(CheckpointError::TimedOut {
|
||||
action: action.to_string(),
|
||||
timeout: self.timeout,
|
||||
}),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The tail of git's stderr for an error message: what the run's record
|
||||
/// carries about the failure, bounded.
|
||||
fn detail(stderr: &[u8]) -> String {
|
||||
const LIMIT: usize = 512;
|
||||
let text = String::from_utf8_lossy(stderr);
|
||||
let text = text.trim();
|
||||
if text.is_empty() {
|
||||
return "no output".to_string();
|
||||
}
|
||||
let start = text.len().saturating_sub(LIMIT);
|
||||
let start = text
|
||||
.char_indices()
|
||||
.map(|(index, _)| index)
|
||||
.find(|index| *index >= start)
|
||||
.unwrap_or(0);
|
||||
text[start..].to_string()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn workspaces(dir: &Path) -> RunWorkspaces {
|
||||
RunWorkspaces::new(
|
||||
dir.to_path_buf(),
|
||||
"run-1".to_string(),
|
||||
GitAuthor::default(),
|
||||
&RunCheckpointSettings::default(),
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_key_round_trips_through_its_ref_and_its_operation() {
|
||||
let key = CheckpointKey {
|
||||
execution: 3,
|
||||
firing: 17,
|
||||
attempt: 2,
|
||||
};
|
||||
assert_eq!(key.snapshot_ref(), "refs/checkpoints/3/17/2");
|
||||
assert_eq!(CheckpointKey::from_ref(&key.snapshot_ref()), Some(key));
|
||||
assert_eq!(CheckpointKey::from_ref("refs/heads/main"), None);
|
||||
assert_eq!(CheckpointKey::from_operation(&key.operation()), Some(key));
|
||||
assert_eq!(
|
||||
CheckpointKey::from_operation(&OperationKey {
|
||||
execution: 3,
|
||||
decision: DecisionRef::Route {
|
||||
firing: 17,
|
||||
attempt: 2,
|
||||
},
|
||||
effect: CHECKPOINT_EFFECT.to_string(),
|
||||
}),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_message_carries_the_identity_as_trailers_last() {
|
||||
let dir = tempfile::tempdir().expect("a temp dir");
|
||||
let key = CheckpointKey {
|
||||
execution: 0,
|
||||
firing: 4,
|
||||
attempt: 1,
|
||||
};
|
||||
let message = workspaces(dir.path()).message(key, "build", "success");
|
||||
assert!(message.starts_with("fabro(run-1): build (success)\n\n"));
|
||||
assert_eq!(CheckpointKey::from_message(&message), Some(key));
|
||||
assert_eq!(trailer::parse(&message, RUN_TRAILER), Some("run-1"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_commit_is_published_found_and_restored() {
|
||||
let dir = tempfile::tempdir().expect("a temp dir");
|
||||
let workspaces = workspaces(dir.path());
|
||||
let workspace = "invocation-0-scope-0";
|
||||
let path = workspaces.workspace_path(workspace);
|
||||
fs::create_dir_all(&path).await.expect("the workspace");
|
||||
fs::write(path.join("out.txt"), "one\n")
|
||||
.await
|
||||
.expect("a file");
|
||||
let key = CheckpointKey {
|
||||
execution: 0,
|
||||
firing: 2,
|
||||
attempt: 1,
|
||||
};
|
||||
|
||||
let first = workspaces
|
||||
.commit(workspace, key, "build", "success")
|
||||
.await
|
||||
.expect("the commit");
|
||||
assert!(!first.reused);
|
||||
let again = workspaces
|
||||
.commit(workspace, key, "build", "success")
|
||||
.await
|
||||
.expect("the second commit");
|
||||
assert_eq!(again, Snapshot {
|
||||
sha: first.sha.clone(),
|
||||
reused: true,
|
||||
});
|
||||
assert_eq!(
|
||||
workspaces.find(workspace, key).await.expect("the lookup"),
|
||||
Some(first.sha.clone())
|
||||
);
|
||||
assert_eq!(
|
||||
workspaces.published(workspace).await.expect("the listing"),
|
||||
vec![PublishedSnapshot {
|
||||
key,
|
||||
sha: first.sha.clone(),
|
||||
}]
|
||||
);
|
||||
assert!(
|
||||
workspaces
|
||||
.matches(workspace, &first.sha)
|
||||
.await
|
||||
.expect("matches")
|
||||
);
|
||||
|
||||
// The stage goes on, then the workspace is lost.
|
||||
fs::write(path.join("out.txt"), "two\n")
|
||||
.await
|
||||
.expect("a change");
|
||||
fs::write(path.join("scratch.txt"), "junk\n")
|
||||
.await
|
||||
.expect("an untracked file");
|
||||
assert!(
|
||||
!workspaces
|
||||
.matches(workspace, &first.sha)
|
||||
.await
|
||||
.expect("matches")
|
||||
);
|
||||
workspaces
|
||||
.reset(workspace, &first.sha)
|
||||
.await
|
||||
.expect("the reset");
|
||||
assert_eq!(
|
||||
fs::read_to_string(path.join("out.txt"))
|
||||
.await
|
||||
.expect("the file"),
|
||||
"one\n"
|
||||
);
|
||||
assert!(
|
||||
!fs::try_exists(path.join("scratch.txt"))
|
||||
.await
|
||||
.expect("exists")
|
||||
);
|
||||
|
||||
fs::remove_dir_all(&path).await.expect("the workspace goes");
|
||||
assert_eq!(
|
||||
workspaces.find(workspace, key).await.expect("the lookup"),
|
||||
Some(first.sha.clone()),
|
||||
"the snapshot repository still knows the commit"
|
||||
);
|
||||
workspaces
|
||||
.restore(workspace, key, &first.sha)
|
||||
.await
|
||||
.expect("the restore");
|
||||
assert_eq!(
|
||||
fs::read_to_string(path.join("out.txt"))
|
||||
.await
|
||||
.expect("the restored file"),
|
||||
"one\n"
|
||||
);
|
||||
assert_eq!(
|
||||
workspaces
|
||||
.workspace_head(workspace)
|
||||
.await
|
||||
.expect("the head"),
|
||||
Some(first.sha)
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn an_unusable_repository_fails_the_commit() {
|
||||
let dir = tempfile::tempdir().expect("a temp dir");
|
||||
let workspaces = workspaces(dir.path());
|
||||
let workspace = "invocation-0-scope-0";
|
||||
let path = workspaces.workspace_path(workspace);
|
||||
fs::create_dir_all(&path).await.expect("the workspace");
|
||||
fs::write(path.join(".git"), "garbage\n")
|
||||
.await
|
||||
.expect("a broken gitfile");
|
||||
let key = CheckpointKey {
|
||||
execution: 0,
|
||||
firing: 2,
|
||||
attempt: 1,
|
||||
};
|
||||
let error = workspaces
|
||||
.commit(workspace, key, "build", "success")
|
||||
.await
|
||||
.expect_err("the commit fails");
|
||||
assert!(matches!(error, CheckpointError::Command { .. }), "{error}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detail_keeps_the_tail_of_long_output() {
|
||||
let long = "x".repeat(600);
|
||||
assert_eq!(detail(long.as_bytes()).len(), 512);
|
||||
assert_eq!(detail(b""), "no output");
|
||||
}
|
||||
}
|
||||
|
|
@ -18,17 +18,20 @@
|
|||
//! What the caller supplies beyond the runtime: the interviewer its
|
||||
//! questions go to ([`interview`](crate::interview) in the worker and the
|
||||
//! server), the secret provider over the vault ([`secrets`](crate::secrets))
|
||||
//! and the blob table ([`blobs`](crate::blobs)) when it has them. What the
|
||||
//! standalone runner's defaults give the run: Petri's local hook service
|
||||
//! for `[[run.hooks]]`, no `ExecutionHooks` of Fabro's own, no host tools,
|
||||
//! and `Retention::Always` for every workspace, Fabro's default.
|
||||
//! and the blob table ([`blobs`](crate::blobs)) when it has them, and a
|
||||
//! [`HooksSpec`] for Fabro's own [`FabroHooks`], which wrap Petri's local
|
||||
//! hook service: the checkpoint commit before every durable finish and its
|
||||
//! platform record after every route, with a failed commit ending the run
|
||||
//! as a `checkpoint_failed` failure. What the standalone runner's defaults
|
||||
//! give the run: Petri's local hook service for `[[run.hooks]]`, no host
|
||||
//! tools, and `Retention::Always` for every workspace, Fabro's default.
|
||||
//! Cancellation rides the caller's token: when it fires, the root
|
||||
//! invocation is cancelled politely and Petri records why.
|
||||
//!
|
||||
//! A resume here is Petri's own: the run continues from its records, and
|
||||
//! sandbox leases are reconciled by label. Full recovery, where the
|
||||
//! workspace a resumed stage sees is restored to the snapshot its durable
|
||||
//! state names, is the integration plan's F3.5 and lands after this.
|
||||
//! sandbox leases are reconciled by label. What the workspaces look like
|
||||
//! when it does is the server's business before it relaunches the worker
|
||||
//! ([`recovery`](crate::recovery)).
|
||||
//!
|
||||
//! No stage or agent event is projected into Fabro's tables here; the
|
||||
//! caller appends only the run lifecycle events Fabro's read side needs to
|
||||
|
|
@ -38,13 +41,14 @@
|
|||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
|
||||
use fabro_types::{FailureReason, SandboxProviderKind};
|
||||
use fabro_types::{FailureReason, RunId, SandboxProviderKind};
|
||||
use petri_execution::host::{self, HostError, HostRun};
|
||||
use petri_execution::inspect::{self, InspectError, RunInspection};
|
||||
use petri_execution::{
|
||||
Access, CancelReason, ExecutionObserver, InterviewDispatcher, Interviewer, InvocationId,
|
||||
RECEIPT_FILE, RunKey, RunStore,
|
||||
};
|
||||
use petri_runtime::driver::lifecycle::ExecutionHooks;
|
||||
use petri_runtime::executor::{Retention, SecretProvider};
|
||||
use petri_runtime::{RunOptions, SandboxBackend};
|
||||
use tokio::fs;
|
||||
|
|
@ -53,6 +57,7 @@ use tracing::{debug, info, warn};
|
|||
|
||||
use crate::admission::AdmittedGraphs;
|
||||
use crate::blobs::{Blobs, RunBlobs};
|
||||
use crate::hooks::{FabroHooks, HooksSpec};
|
||||
use crate::runtime::RuntimeSpec;
|
||||
use crate::secrets::SharedSecrets;
|
||||
|
||||
|
|
@ -95,6 +100,9 @@ pub struct RunRequest {
|
|||
/// Where offloaded stage values go; `None` keeps Petri's local store
|
||||
/// under the run directory.
|
||||
pub blobs: Option<Arc<dyn Blobs>>,
|
||||
/// Fabro's hooks: the checkpoint commit and its record. `None` runs
|
||||
/// with Petri's local hook service alone.
|
||||
pub hooks: Option<HooksSpec>,
|
||||
}
|
||||
|
||||
/// The recorded status of a finished run.
|
||||
|
|
@ -168,12 +176,32 @@ pub async fn run(request: RunRequest) -> Result<RunOutcome, RunError> {
|
|||
if let Some(blobs) = request.blobs {
|
||||
runtime = runtime.capability(RunBlobs::output_store(blobs));
|
||||
}
|
||||
let fabro_hooks = request.hooks.map(|spec| {
|
||||
let inner = runtime
|
||||
.installed_hooks()
|
||||
.unwrap_or_else(|| Arc::new(NoHooks));
|
||||
let run_id = spec_run_id(&request.run_id);
|
||||
Arc::new(FabroHooks::new(
|
||||
spec,
|
||||
inner,
|
||||
run_id,
|
||||
key.clone(),
|
||||
request.run_dir.clone(),
|
||||
Arc::clone(&request.store),
|
||||
))
|
||||
});
|
||||
if let Some(hooks) = &fabro_hooks {
|
||||
runtime = runtime.hooks(Arc::clone(hooks) as Arc<dyn ExecutionHooks>);
|
||||
}
|
||||
|
||||
let dispatcher = InterviewDispatcher::new(request.interviewer);
|
||||
let cancel = request.cancel.clone();
|
||||
let mut cancel_task = None;
|
||||
let with_handle = |handle: petri_execution::CoordinatorHandle, secrets| {
|
||||
dispatcher.wire(handle.clone(), secrets);
|
||||
if let Some(hooks) = &fabro_hooks {
|
||||
hooks.attach(handle.clone());
|
||||
}
|
||||
cancel_task = Some(tokio::spawn(async move {
|
||||
cancel.cancelled().await;
|
||||
info!("cancelling the Petri run");
|
||||
|
|
@ -213,9 +241,37 @@ pub async fn run(request: RunRequest) -> Result<RunOutcome, RunError> {
|
|||
Err(error) => warn!(error = %error, "Petri run ended with a host error"),
|
||||
}
|
||||
let inspection = inspect(request.store.as_ref(), &key).await?;
|
||||
outcome(inspection, result.err())
|
||||
let mut outcome = outcome(inspection, result.err())?;
|
||||
// A failed checkpoint cancelled the run; what Fabro reports is the
|
||||
// checkpoint failure, not a cancellation.
|
||||
if let Some(failure) = fabro_hooks
|
||||
.as_ref()
|
||||
.and_then(|hooks| hooks.checkpoint_failure())
|
||||
{
|
||||
outcome.status = RunStatus::Failed;
|
||||
outcome.failure = Some(failure);
|
||||
}
|
||||
Ok(outcome)
|
||||
}
|
||||
|
||||
/// The Fabro run id the run key names. A key that is not one (a test's
|
||||
/// bare key) still gets hooks, under a fresh id for its platform records.
|
||||
fn spec_run_id(run_id: &str) -> RunId {
|
||||
run_id.parse().unwrap_or_else(|_| {
|
||||
warn!(
|
||||
run_id,
|
||||
"the Petri run key is not a Fabro run id; platform records use a fresh id"
|
||||
);
|
||||
RunId::new()
|
||||
})
|
||||
}
|
||||
|
||||
/// No host hooks at all: what Fabro's hooks wrap when the runtime installed
|
||||
/// none.
|
||||
struct NoHooks;
|
||||
|
||||
impl ExecutionHooks for NoHooks {}
|
||||
|
||||
/// What the run's record says, read through a handle that holds no lease:
|
||||
/// the same derivation [`run`] ends with, for a caller that only holds the
|
||||
/// store, such as a test checking a finished run.
|
||||
|
|
|
|||
583
lib/components/fabro-petri/src/hooks.rs
Normal file
583
lib/components/fabro-petri/src/hooks.rs
Normal file
|
|
@ -0,0 +1,583 @@
|
|||
//! Fabro's awaited extension points on a Petri run: the checkpoint commit,
|
||||
//! its platform record, and the run-level ends, wrapped around Petri's own
|
||||
//! hook service so `[[run.hooks]]` keep running.
|
||||
//!
|
||||
//! [`FabroHooks`] implements Petri's `ExecutionHooks` and is installed with
|
||||
//! `Runtime::hooks` by [`engine::run`](crate::engine::run). It holds the
|
||||
//! hooks the runtime installed before it (Petri's local hook service behind
|
||||
//! its adapter, which serves `[[run.hooks]]`) and forwards every point to
|
||||
//! them, `run_finished` and `scope_released` included, the way Petri's
|
||||
//! embedding host does. Its own work, at each point:
|
||||
//!
|
||||
//! - `prepare_result`: the checkpoint commit, before the `StepFinished` record
|
||||
//! is appended, so a durable finish implies a durable snapshot. A stage that
|
||||
//! failed on its own terms is committed like a successful one; only a
|
||||
//! cancelled attempt is not. A failed commit is fatal to the run: the outcome
|
||||
//! becomes a failure of class `checkpoint_failed`, the run is cancelled
|
||||
//! through the coordinator handle, and `transition` refuses the firing's
|
||||
//! routes, so no route is taken.
|
||||
//! - `transition`: the platform checkpoint record, keyed on the Petri position
|
||||
//! and the checkpoint's operation identity. A failed write is a recorded
|
||||
//! problem on the transition, never a blocked route.
|
||||
//! - `run_finished` and `scope_released`: forwarded, so the local service runs
|
||||
//! `run_complete`, `run_failed` and `sandbox_cleanup` with the sandbox in
|
||||
//! place. Fabro's own end-of-run work (the terminal lifecycle event,
|
||||
//! notifications on it) is the run lifecycle path's, on the worker's and
|
||||
//! server's side of the engine, and the workspace's retention is Petri's
|
||||
//! (`Retention::Always`).
|
||||
//!
|
||||
//! # Operation identities
|
||||
//!
|
||||
//! Every external effect here is keyed on `(run key, execution, DecisionId,
|
||||
//! effect kind)` from the hook context and deduplicated on retry: the
|
||||
//! checkpoint's key is the attempt's decision in its execution, effect
|
||||
//! `checkpoint`. A re-dispatched attempt whose commit already landed
|
||||
//! reuses it when the workspace still sits on it unchanged (see
|
||||
//! [`RunWorkspaces::commit`]); a reissued routing decision finds the
|
||||
//! record, or the commit by its trailers, and writes nothing twice.
|
||||
//!
|
||||
//! # Where the workspace is
|
||||
//!
|
||||
//! The commit runs on the host, in the workspace Petri's host backend keeps
|
||||
//! under the run directory (`crate::checkpoint`). A run on Docker or
|
||||
//! Daytona has no workspace this process can reach; its hooks record that
|
||||
//! no snapshot was taken and leave the run to continue as before.
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::path::PathBuf;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex, MutexGuard, OnceLock, PoisonError};
|
||||
use std::time::Duration;
|
||||
|
||||
use fabro_checkpoint::author::GitAuthor;
|
||||
use fabro_store::platform_records::CheckpointRecord;
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
|
||||
use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace};
|
||||
use fabro_types::{RunId, SandboxProviderKind};
|
||||
use fabro_util::error::collect_chain;
|
||||
use petri_execution::{CancelReason, CoordinatorHandle, InvocationId, RunKey, RunStore};
|
||||
use petri_runtime::driver::lifecycle::{
|
||||
AdmitAttempt, AttemptDecision, ExecutionHooks, HookContext, Note, PrepareError, PrepareResult,
|
||||
Prepared, Recorded, ResultOrigin, RunFinished, ScopeReleased, Transition, TransitionError,
|
||||
TransitionReport,
|
||||
};
|
||||
use petri_runtime::ir::{FailureInfo, ScopeId, Status};
|
||||
use serde_json::json;
|
||||
use tokio::sync::{Mutex as AsyncMutex, OnceCell};
|
||||
use tokio::{fs, time};
|
||||
use tracing::{debug, info, warn};
|
||||
|
||||
use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces};
|
||||
use crate::platform_records::PlatformRecords;
|
||||
use crate::workspace::{self, WorkspaceLookup};
|
||||
|
||||
/// The note kind the hooks record on a firing about its checkpoint.
|
||||
pub const CHECKPOINT_NOTE: &str = "fabro.checkpoint";
|
||||
|
||||
/// How often a held checkpoint polls its test gate.
|
||||
const GATE_POLL: Duration = Duration::from_millis(50);
|
||||
|
||||
/// What Fabro's hooks need beside the run: where the platform records go,
|
||||
/// who authors the commits, and the checkpoint settings.
|
||||
pub struct HooksSpec {
|
||||
pub records: Arc<dyn PlatformRecords>,
|
||||
pub author: GitAuthor,
|
||||
pub checkpoint: RunCheckpointSettings,
|
||||
/// Whether the run's workspaces are on this host (the local sandbox
|
||||
/// provider). A run elsewhere takes no snapshot.
|
||||
pub host_workspaces: bool,
|
||||
/// A test's gate directory: a checkpoint point named by a `.hold` file
|
||||
/// there waits for its `.release` file. `None` outside tests.
|
||||
pub test_gates: Option<PathBuf>,
|
||||
}
|
||||
|
||||
impl HooksSpec {
|
||||
/// The spec a run's settings give: its Git author, its checkpoint
|
||||
/// settings, and whether its sandbox provider keeps workspaces on this
|
||||
/// host.
|
||||
#[must_use]
|
||||
pub fn for_run(records: Arc<dyn PlatformRecords>, settings: &RunNamespace) -> Self {
|
||||
Self {
|
||||
records,
|
||||
author: settings
|
||||
.git
|
||||
.author
|
||||
.as_ref()
|
||||
.map(GitAuthor::from)
|
||||
.unwrap_or_default(),
|
||||
checkpoint: settings.checkpoint.clone(),
|
||||
host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL,
|
||||
test_gates: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[must_use]
|
||||
pub fn with_test_gates(mut self, gates: Option<PathBuf>) -> Self {
|
||||
self.test_gates = gates;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// Fabro's `ExecutionHooks`, around the hooks the runtime installed.
|
||||
pub struct FabroHooks {
|
||||
inner: Arc<dyn ExecutionHooks>,
|
||||
run_id: RunId,
|
||||
records: Arc<dyn PlatformRecords>,
|
||||
workspaces: RunWorkspaces,
|
||||
lookup: WorkspaceLookup,
|
||||
host_workspaces: bool,
|
||||
test_gates: Option<PathBuf>,
|
||||
handle: OnceLock<CoordinatorHandle>,
|
||||
/// The workspace and commit of every checkpoint this process made.
|
||||
committed: Mutex<HashMap<CheckpointKey, (String, String)>>,
|
||||
/// Which checkpoints have their platform record, loaded from the store
|
||||
/// once and kept up to date with every append.
|
||||
recorded: Mutex<HashSet<CheckpointKey>>,
|
||||
recorded_loaded: OnceCell<()>,
|
||||
/// Inherited workspaces resolved through the run's records.
|
||||
inherited: Mutex<HashMap<InvocationId, Option<String>>>,
|
||||
/// One lock per workspace: the branches of a parallel node and a nested
|
||||
/// invocation share their caller's workspace, and Git allows one index
|
||||
/// operation at a time in it.
|
||||
workspace_locks: Mutex<HashMap<String, Arc<AsyncMutex<()>>>>,
|
||||
/// The checkpoint failure that ended the run, when one did.
|
||||
failure: Mutex<Option<String>>,
|
||||
unreachable_noted: AtomicBool,
|
||||
}
|
||||
|
||||
impl FabroHooks {
|
||||
/// Wrap `inner` (the hooks `Runtime::installed_hooks` returned) for the
|
||||
/// run whose records are in `store` under `run_key`, with its
|
||||
/// workspaces under `run_dir`.
|
||||
#[must_use]
|
||||
pub fn new(
|
||||
spec: HooksSpec,
|
||||
inner: Arc<dyn ExecutionHooks>,
|
||||
run_id: RunId,
|
||||
run_key: RunKey,
|
||||
run_dir: PathBuf,
|
||||
store: Arc<dyn RunStore>,
|
||||
) -> Self {
|
||||
let workspaces =
|
||||
RunWorkspaces::new(run_dir, run_id.to_string(), spec.author, &spec.checkpoint);
|
||||
Self {
|
||||
inner,
|
||||
run_id,
|
||||
records: spec.records,
|
||||
workspaces,
|
||||
lookup: WorkspaceLookup::new(store, run_key),
|
||||
host_workspaces: spec.host_workspaces,
|
||||
test_gates: spec.test_gates,
|
||||
handle: OnceLock::new(),
|
||||
committed: Mutex::default(),
|
||||
recorded: Mutex::default(),
|
||||
recorded_loaded: OnceCell::new(),
|
||||
inherited: Mutex::default(),
|
||||
workspace_locks: Mutex::default(),
|
||||
failure: Mutex::default(),
|
||||
unreachable_noted: AtomicBool::new(false),
|
||||
}
|
||||
}
|
||||
|
||||
/// Hand the hooks the running coordinator, so a fatal checkpoint can
|
||||
/// cancel the run. Called once, from the host's handle callback.
|
||||
pub fn attach(&self, handle: CoordinatorHandle) {
|
||||
if self.handle.set(handle).is_err() {
|
||||
debug!("the coordinator handle was already attached to the hooks");
|
||||
}
|
||||
}
|
||||
|
||||
/// The checkpoint failure that ended the run, when one did: what the
|
||||
/// engine reports the run failed with.
|
||||
#[must_use]
|
||||
pub fn checkpoint_failure(&self) -> Option<String> {
|
||||
lock(&self.failure).clone()
|
||||
}
|
||||
|
||||
/// The run's workspaces on this host, as the hooks reach them.
|
||||
#[must_use]
|
||||
pub fn workspaces(&self) -> &RunWorkspaces {
|
||||
&self.workspaces
|
||||
}
|
||||
|
||||
/// The lock that serializes Git work in one workspace.
|
||||
fn workspace_lock(&self, workspace: &str) -> Arc<AsyncMutex<()>> {
|
||||
Arc::clone(
|
||||
lock(&self.workspace_locks)
|
||||
.entry(workspace.to_string())
|
||||
.or_default(),
|
||||
)
|
||||
}
|
||||
|
||||
fn fail_run(&self, message: &str) {
|
||||
let mut failure = lock(&self.failure);
|
||||
if failure.is_none() {
|
||||
*failure = Some(message.to_string());
|
||||
}
|
||||
drop(failure);
|
||||
if let Some(handle) = self.handle.get() {
|
||||
info!(run_id = %self.run_id, "cancelling the Petri run after a failed checkpoint");
|
||||
handle.cancel_root_for(CancelReason::Control);
|
||||
} else {
|
||||
warn!(
|
||||
run_id = %self.run_id,
|
||||
"no coordinator handle is attached; the failed checkpoint cannot cancel the run"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The workspace id of `scope` in the context's invocation: the
|
||||
/// isolated name when its workspace exists, else the inherited one the
|
||||
/// records name, else the isolated name for the caller to report.
|
||||
async fn workspace_of(&self, context: &HookContext, scope: ScopeId) -> Result<String, String> {
|
||||
let isolated = workspace::isolated_workspace(context.invocation, scope);
|
||||
if self.workspaces.workspace_exists(&isolated).await {
|
||||
return Ok(isolated);
|
||||
}
|
||||
let cached = lock(&self.inherited).get(&context.invocation).cloned();
|
||||
let inherited = if let Some(inherited) = cached {
|
||||
inherited
|
||||
} else {
|
||||
let inherited = self
|
||||
.lookup
|
||||
.inherited(context.invocation)
|
||||
.await
|
||||
.map_err(|error| {
|
||||
format!(
|
||||
"the workspace of scope {scope} in invocation {} could not be found: {}",
|
||||
context.invocation,
|
||||
collect_chain(&error).join(": ")
|
||||
)
|
||||
})?;
|
||||
lock(&self.inherited).insert(context.invocation, inherited.clone());
|
||||
inherited
|
||||
};
|
||||
Ok(inherited.unwrap_or(isolated))
|
||||
}
|
||||
|
||||
/// The checkpoint commit for one attempt's result. `Ok(Some)` is the
|
||||
/// note to record, `Ok(None)` nothing to record, `Err` the fatal
|
||||
/// failure message.
|
||||
async fn snapshot(
|
||||
&self,
|
||||
context: &HookContext,
|
||||
scope: ScopeId,
|
||||
key: CheckpointKey,
|
||||
node: &str,
|
||||
status: &Status,
|
||||
origin: ResultOrigin,
|
||||
) -> Result<Option<Note>, String> {
|
||||
if !self.host_workspaces {
|
||||
if !self.unreachable_noted.swap(true, Ordering::SeqCst) {
|
||||
warn!(
|
||||
run_id = %self.run_id,
|
||||
"the run's workspaces are not on this host; no checkpoint snapshot is taken"
|
||||
);
|
||||
}
|
||||
return Ok(Some(Note::new(
|
||||
CHECKPOINT_NOTE,
|
||||
json!({
|
||||
"execution": key.execution,
|
||||
"firing": key.firing,
|
||||
"attempt": key.attempt,
|
||||
"skipped": "the workspace is not on this host",
|
||||
}),
|
||||
)));
|
||||
}
|
||||
let workspace = self.workspace_of(context, scope).await?;
|
||||
if !self.workspaces.workspace_exists(&workspace).await {
|
||||
// A skipped node or a driver-made outcome may precede the scope's
|
||||
// environment; nothing of the stage's is on disk to snapshot.
|
||||
if origin == ResultOrigin::Driver || matches!(status, Status::Skipped) {
|
||||
return Ok(Some(Note::new(
|
||||
CHECKPOINT_NOTE,
|
||||
json!({
|
||||
"execution": key.execution,
|
||||
"firing": key.firing,
|
||||
"attempt": key.attempt,
|
||||
"workspace": workspace,
|
||||
"skipped": "the workspace does not exist yet",
|
||||
}),
|
||||
)));
|
||||
}
|
||||
return Err(format!(
|
||||
"the workspace `{workspace}` of scope {scope} does not exist at {}",
|
||||
self.workspaces.workspace_path(&workspace).display()
|
||||
));
|
||||
}
|
||||
self.gate("commit", node).await;
|
||||
let serialized = self.workspace_lock(&workspace);
|
||||
let _held = serialized.lock().await;
|
||||
match self
|
||||
.workspaces
|
||||
.commit(&workspace, key, node, status.tag())
|
||||
.await
|
||||
{
|
||||
Ok(snapshot) => {
|
||||
debug!(
|
||||
run_id = %self.run_id,
|
||||
node,
|
||||
execution = key.execution,
|
||||
firing = key.firing,
|
||||
attempt = key.attempt,
|
||||
reused = snapshot.reused,
|
||||
"checkpoint committed"
|
||||
);
|
||||
lock(&self.committed).insert(key, (workspace.clone(), snapshot.sha.clone()));
|
||||
Ok(Some(Note::new(
|
||||
CHECKPOINT_NOTE,
|
||||
json!({
|
||||
"execution": key.execution,
|
||||
"firing": key.firing,
|
||||
"attempt": key.attempt,
|
||||
"workspace": workspace,
|
||||
"git_commit_sha": snapshot.sha,
|
||||
"reused": snapshot.reused,
|
||||
}),
|
||||
)))
|
||||
}
|
||||
Err(error) => Err(format!(
|
||||
"the checkpoint commit of `{node}` failed: {}",
|
||||
collect_chain(&error).join(": ")
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
/// The checkpoint's platform record, once per operation identity.
|
||||
async fn record(
|
||||
&self,
|
||||
context: &HookContext,
|
||||
scope: ScopeId,
|
||||
key: CheckpointKey,
|
||||
) -> Result<(), String> {
|
||||
self.recorded_loaded
|
||||
.get_or_try_init(|| self.load_recorded())
|
||||
.await?;
|
||||
if lock(&self.recorded).contains(&key) {
|
||||
return Ok(());
|
||||
}
|
||||
let committed = lock(&self.committed).get(&key).cloned();
|
||||
let (workspace, sha) = if let Some(committed) = committed {
|
||||
committed
|
||||
} else {
|
||||
let workspace = self.workspace_of(context, scope).await?;
|
||||
let serialized = self.workspace_lock(&workspace);
|
||||
let held = serialized.lock().await;
|
||||
let found = self.workspaces.find(&workspace, key).await;
|
||||
drop(held);
|
||||
let sha = found
|
||||
.map_err(|error| {
|
||||
format!(
|
||||
"the checkpoint commit could not be looked up: {}",
|
||||
collect_chain(&error).join(": ")
|
||||
)
|
||||
})?
|
||||
.ok_or_else(|| {
|
||||
format!(
|
||||
"no checkpoint commit exists for execution {} firing {} attempt {}",
|
||||
key.execution, key.firing, key.attempt
|
||||
)
|
||||
})?;
|
||||
(workspace, sha)
|
||||
};
|
||||
let record = PlatformRecord::Checkpoint(CheckpointRecord {
|
||||
execution: key.execution,
|
||||
firing: key.firing,
|
||||
attempt: Some(key.attempt),
|
||||
workspace: Some(workspace),
|
||||
git_commit_sha: Some(sha),
|
||||
diff_summary: None,
|
||||
patch_blob: None,
|
||||
operation: Some(key.operation()),
|
||||
});
|
||||
self.records
|
||||
.append(
|
||||
&self.run_id,
|
||||
&record,
|
||||
Some(StagePosition {
|
||||
execution: key.execution,
|
||||
firing: key.firing,
|
||||
}),
|
||||
)
|
||||
.await
|
||||
.map_err(|error| {
|
||||
format!(
|
||||
"the checkpoint record could not be written: {}",
|
||||
collect_chain(&error).join(": ")
|
||||
)
|
||||
})?;
|
||||
lock(&self.recorded).insert(key);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The checkpoints already recorded for the run, read once: what a
|
||||
/// resume's reissued routing decisions must not record again.
|
||||
async fn load_recorded(&self) -> Result<(), String> {
|
||||
let stored = self
|
||||
.records
|
||||
.read_kind(&self.run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.map_err(|error| {
|
||||
format!(
|
||||
"the run's checkpoint records could not be read: {}",
|
||||
collect_chain(&error).join(": ")
|
||||
)
|
||||
})?;
|
||||
let mut recorded = lock(&self.recorded);
|
||||
for record in stored {
|
||||
let PlatformRecord::Checkpoint(checkpoint) = &record.record else {
|
||||
continue;
|
||||
};
|
||||
if let Some(key) = checkpoint
|
||||
.operation
|
||||
.as_ref()
|
||||
.and_then(CheckpointKey::from_operation)
|
||||
{
|
||||
recorded.insert(key);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Hold at a test gate when one is set for this point and node.
|
||||
async fn gate(&self, point: &str, node: &str) {
|
||||
let Some(dir) = &self.test_gates else {
|
||||
return;
|
||||
};
|
||||
let hold = dir.join(format!("{point}.{node}.hold"));
|
||||
if !fs::try_exists(&hold).await.unwrap_or(false) {
|
||||
return;
|
||||
}
|
||||
let release = dir.join(format!("{point}.{node}.release"));
|
||||
info!(point, node, "checkpoint held at a test gate");
|
||||
while !fs::try_exists(&release).await.unwrap_or(false) {
|
||||
time::sleep(GATE_POLL).await;
|
||||
}
|
||||
info!(point, node, "checkpoint released by its test gate");
|
||||
}
|
||||
}
|
||||
|
||||
fn lock<T>(mutex: &Mutex<T>) -> MutexGuard<'_, T> {
|
||||
mutex.lock().unwrap_or_else(PoisonError::into_inner)
|
||||
}
|
||||
|
||||
fn is_checkpoint_failure(status: &Status) -> bool {
|
||||
matches!(status, Status::Failure(info) if info.class.as_str() == CHECKPOINT_FAILED_CLASS)
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl ExecutionHooks for FabroHooks {
|
||||
async fn before_attempt(
|
||||
&self,
|
||||
context: &HookContext,
|
||||
request: AdmitAttempt,
|
||||
) -> AttemptDecision {
|
||||
self.inner.before_attempt(context, request).await
|
||||
}
|
||||
|
||||
async fn prepare_result(
|
||||
&self,
|
||||
context: &HookContext,
|
||||
request: PrepareResult,
|
||||
) -> Result<Prepared, PrepareError> {
|
||||
let node = request.view.node_name().to_owned();
|
||||
let scope = request.view.scope;
|
||||
let key = CheckpointKey {
|
||||
execution: context.execution.raw(),
|
||||
firing: request.view.firing.raw(),
|
||||
attempt: request.view.attempt.raw(),
|
||||
};
|
||||
let original = request.outcome.status.clone();
|
||||
let origin = request.origin;
|
||||
let mut prepared = self.inner.prepare_result(context, request).await?;
|
||||
let effective = prepared.adjustment.status.clone().unwrap_or(original);
|
||||
if matches!(effective, Status::Cancelled) {
|
||||
return Ok(prepared);
|
||||
}
|
||||
match self
|
||||
.snapshot(context, scope, key, &node, &effective, origin)
|
||||
.await
|
||||
{
|
||||
Ok(Some(note)) => prepared.notes.push(note),
|
||||
Ok(None) => {}
|
||||
Err(message) => {
|
||||
warn!(
|
||||
run_id = %self.run_id,
|
||||
node,
|
||||
execution = key.execution,
|
||||
firing = key.firing,
|
||||
attempt = key.attempt,
|
||||
error = %message,
|
||||
"checkpoint failed; the run ends"
|
||||
);
|
||||
self.fail_run(&message);
|
||||
prepared.adjustment.status = Some(Status::Failure(
|
||||
FailureInfo::new(message.clone()).with_class(CHECKPOINT_FAILED_CLASS),
|
||||
));
|
||||
prepared.adjustment.reason = Some(message);
|
||||
}
|
||||
}
|
||||
Ok(prepared)
|
||||
}
|
||||
|
||||
async fn after_record(&self, context: &HookContext, recorded: Recorded) -> Vec<Note> {
|
||||
self.inner.after_record(context, recorded).await
|
||||
}
|
||||
|
||||
async fn transition(
|
||||
&self,
|
||||
context: &HookContext,
|
||||
transition: Transition,
|
||||
) -> Result<TransitionReport, TransitionError> {
|
||||
if is_checkpoint_failure(&transition.outcome.status) {
|
||||
return Err(TransitionError::new(
|
||||
"the stage's checkpoint commit failed; no route is taken",
|
||||
));
|
||||
}
|
||||
let node = transition.view.node_name().to_owned();
|
||||
let scope = transition.view.scope;
|
||||
let key = CheckpointKey {
|
||||
execution: context.execution.raw(),
|
||||
firing: transition.view.firing.raw(),
|
||||
attempt: transition.view.attempt.raw(),
|
||||
};
|
||||
let mut problems = Vec::new();
|
||||
if self.host_workspaces {
|
||||
self.gate("record", &node).await;
|
||||
if let Err(problem) = self.record(context, scope, key).await {
|
||||
warn!(
|
||||
run_id = %self.run_id,
|
||||
node,
|
||||
execution = key.execution,
|
||||
firing = key.firing,
|
||||
error = %problem,
|
||||
"the checkpoint record was not written"
|
||||
);
|
||||
problems.push(problem);
|
||||
}
|
||||
}
|
||||
let mut report = self.inner.transition(context, transition).await?;
|
||||
report.problems.extend(problems);
|
||||
Ok(report)
|
||||
}
|
||||
|
||||
async fn run_finished(&self, context: &HookContext, finished: RunFinished) -> Vec<Note> {
|
||||
info!(
|
||||
run_id = %self.run_id,
|
||||
status = ?finished.status,
|
||||
failure = finished.failure.as_deref().unwrap_or(""),
|
||||
"Petri run finished; running the run-end hooks"
|
||||
);
|
||||
self.inner.run_finished(context, finished).await
|
||||
}
|
||||
|
||||
async fn scope_released(&self, context: &HookContext, released: ScopeReleased) -> Vec<Note> {
|
||||
debug!(
|
||||
run_id = %self.run_id,
|
||||
scope = %released.scope,
|
||||
outcome = ?released.outcome,
|
||||
"scope released; running the sandbox cleanup hooks"
|
||||
);
|
||||
self.inner.scope_released(context, released).await
|
||||
}
|
||||
}
|
||||
|
|
@ -33,7 +33,17 @@
|
|||
//! - [`projection`] and [`projector`]: the view of a Petri run, folded from its
|
||||
//! records and Fabro's platform records, and the pass that writes it after
|
||||
//! each committed record;
|
||||
//! - the platform adapters still to come: hooks and the run tools.
|
||||
//! - [`hooks`]: Fabro's `ExecutionHooks`, the checkpoint commit in
|
||||
//! `prepare_result` and its platform record in `transition`, around Petri's
|
||||
//! own hook service for `[[run.hooks]]`;
|
||||
//! - [`checkpoint`]: the Git snapshots of a run's host workspaces and the
|
||||
//! snapshot repository they are published to;
|
||||
//! - [`recovery`]: the resume-on-restart protocol, which brings every live
|
||||
//! workspace to the snapshot its durable state names before the run goes back
|
||||
//! to a worker;
|
||||
//! - [`platform_records`]: Fabro's platform records as the adapters reach them,
|
||||
//! in the server's database or over its API from a worker;
|
||||
//! - the platform adapter still to come: the run tools.
|
||||
//!
|
||||
//! The Petri packages are pinned by revision in the workspace `Cargo.toml`
|
||||
//! under `petri_*` keys.
|
||||
|
|
@ -41,17 +51,22 @@
|
|||
pub mod admission;
|
||||
pub mod blobs;
|
||||
pub mod check;
|
||||
pub mod checkpoint;
|
||||
pub mod engine;
|
||||
pub mod hooks;
|
||||
pub mod http_store;
|
||||
pub mod interview;
|
||||
pub mod petri;
|
||||
pub mod platform_records;
|
||||
pub mod projection;
|
||||
pub mod projector;
|
||||
pub mod recovery;
|
||||
pub mod run_store;
|
||||
pub mod runtime;
|
||||
pub mod secrets;
|
||||
#[cfg(feature = "test-support")]
|
||||
pub mod test_support;
|
||||
pub mod workspace;
|
||||
|
||||
pub use http_store::HttpRunStore;
|
||||
pub use run_store::SqliteRunStore;
|
||||
|
|
|
|||
188
lib/components/fabro-petri/src/platform_records.rs
Normal file
188
lib/components/fabro-petri/src/platform_records.rs
Normal file
|
|
@ -0,0 +1,188 @@
|
|||
//! Fabro's platform records as a Petri run's adapters reach them.
|
||||
//!
|
||||
//! The records themselves are `fabro_store::platform_records`: one table
|
||||
//! beside Petri's records, one typed enum of kinds. What differs is where
|
||||
//! the adapter runs. In the server process (the in-process test path, and
|
||||
//! startup recovery) the table is reached directly, through
|
||||
//! [`SqlitePlatformRecords`]; in a run's worker process it is reached over
|
||||
//! the server's API with the worker's token, through
|
||||
//! [`HttpPlatformRecords`], as the run's Petri records are. Both answer
|
||||
//! the one [`PlatformRecords`] interface the hooks and recovery use.
|
||||
|
||||
use std::fmt;
|
||||
use std::sync::Arc;
|
||||
|
||||
use async_trait::async_trait;
|
||||
use fabro_api::types::{PetriPlatformRecord, PetriPlatformRecordAppendRequest};
|
||||
use fabro_client::Client;
|
||||
use fabro_store::{
|
||||
PlatformRecord, PlatformRecordKind, PlatformRecordStore, RunSummaryStore, StagePosition,
|
||||
StoredPlatformRecord,
|
||||
};
|
||||
use fabro_types::RunId;
|
||||
use serde_json::Value;
|
||||
|
||||
/// Why a platform record could not be stored or read.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum PlatformRecordError {
|
||||
#[error("the platform record store failed")]
|
||||
Store(#[source] fabro_store::Error),
|
||||
#[error("the platform record request to the server failed")]
|
||||
Api(#[source] anyhow::Error),
|
||||
#[error("the platform record does not encode as JSON")]
|
||||
Encode(#[source] serde_json::Error),
|
||||
}
|
||||
|
||||
/// The platform records of a run, wherever the adapter runs.
|
||||
#[async_trait]
|
||||
pub trait PlatformRecords: Send + Sync {
|
||||
/// Store a record at the run's next seq, tied to a Petri stage when it
|
||||
/// belongs to one.
|
||||
async fn append(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
record: &PlatformRecord,
|
||||
position: Option<StagePosition>,
|
||||
) -> Result<StoredPlatformRecord, PlatformRecordError>;
|
||||
|
||||
/// The run's records of one kind, in seq order.
|
||||
async fn read_kind(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
kind: PlatformRecordKind,
|
||||
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError>;
|
||||
}
|
||||
|
||||
/// The table in the server's database, with the projector's wake-up after
|
||||
/// each append.
|
||||
#[derive(Clone)]
|
||||
pub struct SqlitePlatformRecords {
|
||||
store: PlatformRecordStore,
|
||||
summaries: Arc<RunSummaryStore>,
|
||||
}
|
||||
|
||||
impl fmt::Debug for SqlitePlatformRecords {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
f.debug_struct("SqlitePlatformRecords")
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
impl SqlitePlatformRecords {
|
||||
#[must_use]
|
||||
pub fn new(summaries: Arc<RunSummaryStore>) -> Self {
|
||||
Self {
|
||||
store: summaries.platform_records(),
|
||||
summaries,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl PlatformRecords for SqlitePlatformRecords {
|
||||
async fn append(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
record: &PlatformRecord,
|
||||
position: Option<StagePosition>,
|
||||
) -> Result<StoredPlatformRecord, PlatformRecordError> {
|
||||
let stored = self
|
||||
.store
|
||||
.append(run_id, record, position)
|
||||
.await
|
||||
.map_err(PlatformRecordError::Store)?;
|
||||
self.summaries.notify_platform_record(*run_id);
|
||||
Ok(stored)
|
||||
}
|
||||
|
||||
async fn read_kind(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
kind: PlatformRecordKind,
|
||||
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
|
||||
self.store
|
||||
.read_kind(run_id, kind)
|
||||
.await
|
||||
.map_err(PlatformRecordError::Store)
|
||||
}
|
||||
}
|
||||
|
||||
/// The table as a run's worker reaches it: the server's
|
||||
/// `/api/v1/runs/{id}/petri/platform-records` endpoints with the worker's
|
||||
/// token.
|
||||
pub struct HttpPlatformRecords {
|
||||
client: Client,
|
||||
}
|
||||
|
||||
impl fmt::Debug for HttpPlatformRecords {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
f.debug_struct("HttpPlatformRecords")
|
||||
.field("server", &self.client.base_url())
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
impl HttpPlatformRecords {
|
||||
#[must_use]
|
||||
pub fn new(client: Client) -> Self {
|
||||
Self { client }
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl PlatformRecords for HttpPlatformRecords {
|
||||
async fn append(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
record: &PlatformRecord,
|
||||
position: Option<StagePosition>,
|
||||
) -> Result<StoredPlatformRecord, PlatformRecordError> {
|
||||
let Value::Object(record) =
|
||||
serde_json::to_value(record).map_err(PlatformRecordError::Encode)?
|
||||
else {
|
||||
return Err(PlatformRecordError::Api(anyhow::anyhow!(
|
||||
"a platform record encodes as a JSON object"
|
||||
)));
|
||||
};
|
||||
let body = PetriPlatformRecordAppendRequest {
|
||||
record,
|
||||
execution: position.map(|position| position.execution),
|
||||
firing: position.map(|position| position.firing),
|
||||
};
|
||||
let stored = self
|
||||
.client
|
||||
.append_petri_platform_record(run_id, body)
|
||||
.await
|
||||
.map_err(PlatformRecordError::Api)?;
|
||||
decode(stored)
|
||||
}
|
||||
|
||||
async fn read_kind(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
kind: PlatformRecordKind,
|
||||
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
|
||||
let records = self
|
||||
.client
|
||||
.list_petri_platform_records(run_id, Some(&kind.to_string()))
|
||||
.await
|
||||
.map_err(PlatformRecordError::Api)?;
|
||||
records.into_iter().map(decode).collect()
|
||||
}
|
||||
}
|
||||
|
||||
/// A wire record back into the store's shape.
|
||||
fn decode(wire: PetriPlatformRecord) -> Result<StoredPlatformRecord, PlatformRecordError> {
|
||||
let record: PlatformRecord =
|
||||
serde_json::from_value(Value::Object(wire.record)).map_err(PlatformRecordError::Encode)?;
|
||||
let position = match (wire.execution, wire.firing) {
|
||||
(Some(execution), Some(firing)) => Some(StagePosition { execution, firing }),
|
||||
_ => None,
|
||||
};
|
||||
Ok(StoredPlatformRecord {
|
||||
seq: wire.seq,
|
||||
recorded_at: wire.recorded_at,
|
||||
record,
|
||||
position,
|
||||
})
|
||||
}
|
||||
450
lib/components/fabro-petri/src/recovery.rs
Normal file
450
lib/components/fabro-petri/src/recovery.rs
Normal file
|
|
@ -0,0 +1,450 @@
|
|||
//! Resume on restart: the recovery protocol whose rule is that the
|
||||
//! workspace a resumed stage sees matches Petri's durable execution state
|
||||
//! (the integration plan's F3.5).
|
||||
//!
|
||||
//! For a Petri run the server finds in flight at startup, once the previous
|
||||
//! worker's lease is released, [`recover`] reads the durable execution
|
||||
//! state through `inspect_run` and decides:
|
||||
//!
|
||||
//! - a run with a `checkpoint_failed` finish anywhere is reported failed and
|
||||
//! not resumed: a failed checkpoint cancelled it, and nothing of it is
|
||||
//! reconciled;
|
||||
//! - otherwise, for every live execution, the last durable finish names the
|
||||
//! snapshot its workspace must sit on: the checkpoint record's commit, or,
|
||||
//! when the record was lost to the crash, the commit found by its key in the
|
||||
//! workspace's snapshot repository or history, which is then recorded again;
|
||||
//! - a workspace that survives is verified to sit on that commit, unchanged, or
|
||||
//! reset to it; a workspace that is gone is restored from the run's snapshot
|
||||
//! repository into a fresh directory;
|
||||
//! - a durable finish with no snapshot fails the run with a named error rather
|
||||
//! than resume it on stale files.
|
||||
//!
|
||||
//! Every child invocation's scope has its own snapshots, keyed by
|
||||
//! execution; a nested invocation that inherits its caller's sandbox shares
|
||||
//! the caller's workspace, and the workspace is brought to the newest of
|
||||
//! the live executions' snapshots on it.
|
||||
//!
|
||||
//! A run whose workspaces are not on this host (Docker, Daytona) is resumed
|
||||
//! on its retained sandbox as it was left: the snapshot side of the
|
||||
//! protocol reaches only host workspaces.
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
|
||||
use fabro_checkpoint::author::GitAuthor;
|
||||
use fabro_store::platform_records::CheckpointRecord;
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition};
|
||||
use fabro_types::settings::run::{RunCheckpointSettings, RunNamespace};
|
||||
use fabro_types::{RunId, SandboxProviderKind};
|
||||
use petri_execution::host::{self, HostError};
|
||||
use petri_execution::inspect::{self, ExecutionInspection, InspectError};
|
||||
use petri_execution::{Access, InvocationId, RunKey, RunStore};
|
||||
use petri_store::StoreError;
|
||||
use tracing::{info, warn};
|
||||
|
||||
use crate::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointError, CheckpointKey, RunWorkspaces};
|
||||
use crate::platform_records::{PlatformRecordError, PlatformRecords};
|
||||
use crate::workspace::{WorkspaceLookup, WorkspaceLookupError};
|
||||
|
||||
/// What recovery needs: the run, where its workspaces are, its records.
|
||||
pub struct RecoveryRequest {
|
||||
pub run_id: RunId,
|
||||
/// The run directory Petri ran under (the run's `petri` scratch).
|
||||
pub run_dir: PathBuf,
|
||||
pub store: Arc<dyn RunStore>,
|
||||
pub records: Arc<dyn PlatformRecords>,
|
||||
pub author: GitAuthor,
|
||||
pub checkpoint: RunCheckpointSettings,
|
||||
/// Whether the run's workspaces are on this host.
|
||||
pub host_workspaces: bool,
|
||||
}
|
||||
|
||||
impl RecoveryRequest {
|
||||
/// The request a run's settings give: its Git author, its checkpoint
|
||||
/// settings, and whether its sandbox provider keeps workspaces on this
|
||||
/// host.
|
||||
#[must_use]
|
||||
pub fn for_run(
|
||||
run_id: RunId,
|
||||
run_dir: PathBuf,
|
||||
store: Arc<dyn RunStore>,
|
||||
records: Arc<dyn PlatformRecords>,
|
||||
settings: &RunNamespace,
|
||||
) -> Self {
|
||||
Self {
|
||||
run_id,
|
||||
run_dir,
|
||||
store,
|
||||
records,
|
||||
author: settings
|
||||
.git
|
||||
.author
|
||||
.as_ref()
|
||||
.map(GitAuthor::from)
|
||||
.unwrap_or_default(),
|
||||
checkpoint: settings.checkpoint.clone(),
|
||||
host_workspaces: settings.environment.provider == SandboxProviderKind::LOCAL,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// What was done to one workspace.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub enum WorkspaceAction {
|
||||
/// It sat on the snapshot, unchanged.
|
||||
Verified,
|
||||
/// It was brought back to the snapshot.
|
||||
Reset,
|
||||
/// It was gone and was recreated from the snapshot repository.
|
||||
Restored,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct RecoveredWorkspace {
|
||||
pub workspace: String,
|
||||
pub sha: String,
|
||||
pub action: WorkspaceAction,
|
||||
}
|
||||
|
||||
/// What the server does with the run next.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub enum Recovery {
|
||||
/// The store never held the run: it starts from its admitted graphs.
|
||||
Start,
|
||||
/// The run continues from its records, its live workspaces on their
|
||||
/// snapshots.
|
||||
Resume { workspaces: Vec<RecoveredWorkspace> },
|
||||
/// The run cannot continue and is reported failed.
|
||||
Failed { reason: String },
|
||||
}
|
||||
|
||||
/// Why recovery could not decide: the records could not be read, or a
|
||||
/// workspace could not be brought to its snapshot.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum RecoveryError {
|
||||
#[error("the run's record could not be opened")]
|
||||
Open(#[source] StoreError),
|
||||
#[error("the run's coordinator log could not be read")]
|
||||
Log(#[source] petri_execution::StoreError),
|
||||
#[error("the run's coordinator state could not be read")]
|
||||
State(#[source] HostError),
|
||||
#[error("the run's record could not be inspected")]
|
||||
Inspect(#[source] InspectError),
|
||||
#[error("the run's workspaces could not be named")]
|
||||
Lookup(#[source] WorkspaceLookupError),
|
||||
#[error("the run's checkpoint records could not be read or written")]
|
||||
Records(#[source] PlatformRecordError),
|
||||
#[error("the workspace `{workspace}` could not be brought to its snapshot")]
|
||||
Workspace {
|
||||
workspace: String,
|
||||
#[source]
|
||||
source: CheckpointError,
|
||||
},
|
||||
}
|
||||
|
||||
/// One live execution's last durable finish and the snapshot it names.
|
||||
struct Target {
|
||||
execution: u64,
|
||||
key: CheckpointKey,
|
||||
}
|
||||
|
||||
/// Decide how the run continues, and bring its workspaces to their
|
||||
/// snapshots.
|
||||
pub async fn recover(request: RecoveryRequest) -> Result<Recovery, RecoveryError> {
|
||||
let key = RunKey::new(request.run_id.to_string());
|
||||
let logs = match request.store.open(&key, Access::Read).await {
|
||||
Ok(logs) => logs,
|
||||
Err(StoreError::NotFound { .. }) => return Ok(Recovery::Start),
|
||||
Err(error) => return Err(RecoveryError::Open(error)),
|
||||
};
|
||||
// A record with no root invocation (the worker died between creating
|
||||
// the run and declaring it) has nothing to reconcile; the worker's
|
||||
// resume reports it as such.
|
||||
let records = petri_execution::read_coordinator_log(&*logs)
|
||||
.await
|
||||
.map_err(RecoveryError::Log)?;
|
||||
if records.is_empty() {
|
||||
return Ok(Recovery::Resume {
|
||||
workspaces: Vec::new(),
|
||||
});
|
||||
}
|
||||
let state = host::stored_state(&*logs)
|
||||
.await
|
||||
.map_err(RecoveryError::State)?;
|
||||
if !state.invocations.contains_key(&InvocationId::ROOT) {
|
||||
return Ok(Recovery::Resume {
|
||||
workspaces: Vec::new(),
|
||||
});
|
||||
}
|
||||
let inspection = inspect::inspect_run(&*logs)
|
||||
.await
|
||||
.map_err(RecoveryError::Inspect)?;
|
||||
drop(logs);
|
||||
|
||||
if let Some(failed) = checkpoint_failure(&inspection.executions) {
|
||||
return Ok(Recovery::Failed { reason: failed });
|
||||
}
|
||||
if !request.host_workspaces {
|
||||
warn!(
|
||||
run_id = %request.run_id,
|
||||
"the run's workspaces are not on this host; resuming on the retained sandbox as it was left"
|
||||
);
|
||||
return Ok(Recovery::Resume {
|
||||
workspaces: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
let workspaces = RunWorkspaces::new(
|
||||
request.run_dir.clone(),
|
||||
request.run_id.to_string(),
|
||||
request.author.clone(),
|
||||
&request.checkpoint,
|
||||
);
|
||||
let lookup = WorkspaceLookup::new(Arc::clone(&request.store), key);
|
||||
let recorded = recorded_checkpoints(&*request.records, &request.run_id).await?;
|
||||
|
||||
// The snapshot each live execution's workspace must sit on. A live
|
||||
// execution is one whose log records no exit: `inspect_run` reports it
|
||||
// as incomplete.
|
||||
let mut candidates: BTreeMap<String, Vec<(Target, String)>> = BTreeMap::new();
|
||||
for execution in inspection
|
||||
.executions
|
||||
.iter()
|
||||
.filter(|execution| execution.status == "incomplete")
|
||||
{
|
||||
let Some(target) = last_finish(execution) else {
|
||||
continue;
|
||||
};
|
||||
let owned = lookup
|
||||
.of_invocation(execution.invocation)
|
||||
.await
|
||||
.map_err(RecoveryError::Lookup)?;
|
||||
if owned.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let mut found = false;
|
||||
for workspace in owned {
|
||||
let sha = match recorded.get(&target.key) {
|
||||
Some((recorded_workspace, sha))
|
||||
if recorded_workspace.as_deref().is_none_or(|w| w == workspace) =>
|
||||
{
|
||||
Some(sha.clone())
|
||||
}
|
||||
_ => {
|
||||
let sha = workspaces
|
||||
.find(&workspace, target.key)
|
||||
.await
|
||||
.map_err(|source| RecoveryError::Workspace {
|
||||
workspace: workspace.clone(),
|
||||
source,
|
||||
})?;
|
||||
if let Some(sha) = &sha {
|
||||
reconcile_record(
|
||||
&*request.records,
|
||||
&request.run_id,
|
||||
target.key,
|
||||
&workspace,
|
||||
sha,
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
sha
|
||||
}
|
||||
};
|
||||
if let Some(sha) = sha {
|
||||
found = true;
|
||||
candidates
|
||||
.entry(workspace)
|
||||
.or_default()
|
||||
.push((Target { ..target }, sha));
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
return Ok(Recovery::Failed {
|
||||
reason: format!(
|
||||
"no checkpoint snapshot exists for the last durable finish of execution {} \
|
||||
(firing {} attempt {}); the run cannot resume on stale files",
|
||||
target.execution, target.key.firing, target.key.attempt
|
||||
),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let mut recovered = Vec::new();
|
||||
for (workspace, targets) in candidates {
|
||||
let sha = newest(&workspaces, &workspace, &targets).await?;
|
||||
let action = bring_to(&workspaces, &workspace, &sha, &targets).await?;
|
||||
info!(
|
||||
run_id = %request.run_id,
|
||||
workspace,
|
||||
sha,
|
||||
action = ?action,
|
||||
"workspace brought to its durable snapshot"
|
||||
);
|
||||
recovered.push(RecoveredWorkspace {
|
||||
workspace,
|
||||
sha,
|
||||
action,
|
||||
});
|
||||
}
|
||||
Ok(Recovery::Resume {
|
||||
workspaces: recovered,
|
||||
})
|
||||
}
|
||||
|
||||
/// The reason a run with a failed checkpoint is reported failed, when it
|
||||
/// has one.
|
||||
fn checkpoint_failure(executions: &[ExecutionInspection]) -> Option<String> {
|
||||
executions.iter().find_map(|execution| {
|
||||
execution
|
||||
.engine
|
||||
.as_ref()?
|
||||
.attempts
|
||||
.iter()
|
||||
.find_map(|attempt| {
|
||||
let failure = attempt.failure.as_ref()?;
|
||||
(failure.class.as_str() == CHECKPOINT_FAILED_CLASS).then(|| {
|
||||
format!(
|
||||
"the checkpoint of {} (execution {} firing {} attempt {}) failed: {}",
|
||||
attempt.node.as_deref().unwrap_or("a stage"),
|
||||
execution.execution,
|
||||
attempt.firing,
|
||||
attempt.attempt,
|
||||
failure.message
|
||||
)
|
||||
})
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// The last `StepFinished` of an execution's log.
|
||||
fn last_finish(execution: &ExecutionInspection) -> Option<Target> {
|
||||
let attempt = execution.engine.as_ref()?.attempts.last()?;
|
||||
Some(Target {
|
||||
execution: execution.execution.raw(),
|
||||
key: CheckpointKey {
|
||||
execution: execution.execution.raw(),
|
||||
firing: attempt.firing,
|
||||
attempt: attempt.attempt,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
/// The run's checkpoint records by key: the workspace they name and the
|
||||
/// commit.
|
||||
async fn recorded_checkpoints(
|
||||
records: &dyn PlatformRecords,
|
||||
run_id: &RunId,
|
||||
) -> Result<BTreeMap<CheckpointKey, (Option<String>, String)>, RecoveryError> {
|
||||
let stored = records
|
||||
.read_kind(run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.map_err(RecoveryError::Records)?;
|
||||
let mut recorded = BTreeMap::new();
|
||||
for record in stored {
|
||||
let PlatformRecord::Checkpoint(checkpoint) = record.record else {
|
||||
continue;
|
||||
};
|
||||
let key = checkpoint
|
||||
.operation
|
||||
.as_ref()
|
||||
.and_then(CheckpointKey::from_operation);
|
||||
if let (Some(key), Some(sha)) = (key, checkpoint.git_commit_sha) {
|
||||
recorded.insert(key, (checkpoint.workspace, sha));
|
||||
}
|
||||
}
|
||||
Ok(recorded)
|
||||
}
|
||||
|
||||
/// Write the record a crash lost, from the commit found by its key.
|
||||
async fn reconcile_record(
|
||||
records: &dyn PlatformRecords,
|
||||
run_id: &RunId,
|
||||
key: CheckpointKey,
|
||||
workspace: &str,
|
||||
sha: &str,
|
||||
) -> Result<(), RecoveryError> {
|
||||
info!(
|
||||
run_id = %run_id,
|
||||
execution = key.execution,
|
||||
firing = key.firing,
|
||||
attempt = key.attempt,
|
||||
sha,
|
||||
"checkpoint record reconciled from the run branch"
|
||||
);
|
||||
let record = PlatformRecord::Checkpoint(CheckpointRecord {
|
||||
execution: key.execution,
|
||||
firing: key.firing,
|
||||
attempt: Some(key.attempt),
|
||||
workspace: Some(workspace.to_string()),
|
||||
git_commit_sha: Some(sha.to_string()),
|
||||
diff_summary: None,
|
||||
patch_blob: None,
|
||||
operation: Some(key.operation()),
|
||||
});
|
||||
records
|
||||
.append(
|
||||
run_id,
|
||||
&record,
|
||||
Some(StagePosition {
|
||||
execution: key.execution,
|
||||
firing: key.firing,
|
||||
}),
|
||||
)
|
||||
.await
|
||||
.map_err(RecoveryError::Records)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Of the snapshots live executions name on one workspace, the one every
|
||||
/// other descends from, else the last named.
|
||||
async fn newest(
|
||||
workspaces: &RunWorkspaces,
|
||||
workspace: &str,
|
||||
targets: &[(Target, String)],
|
||||
) -> Result<String, RecoveryError> {
|
||||
let mut chosen = &targets[0].1;
|
||||
for (_, sha) in &targets[1..] {
|
||||
if workspaces
|
||||
.is_ancestor(workspace, chosen, sha)
|
||||
.await
|
||||
.map_err(|source| RecoveryError::Workspace {
|
||||
workspace: workspace.to_string(),
|
||||
source,
|
||||
})?
|
||||
{
|
||||
chosen = sha;
|
||||
}
|
||||
}
|
||||
Ok(chosen.clone())
|
||||
}
|
||||
|
||||
/// Verify, reset or restore the workspace onto `sha`.
|
||||
async fn bring_to(
|
||||
workspaces: &RunWorkspaces,
|
||||
workspace: &str,
|
||||
sha: &str,
|
||||
targets: &[(Target, String)],
|
||||
) -> Result<WorkspaceAction, RecoveryError> {
|
||||
let failed = |source| RecoveryError::Workspace {
|
||||
workspace: workspace.to_string(),
|
||||
source,
|
||||
};
|
||||
if workspaces.workspace_exists(workspace).await {
|
||||
if workspaces.matches(workspace, sha).await.map_err(failed)? {
|
||||
return Ok(WorkspaceAction::Verified);
|
||||
}
|
||||
workspaces.reset(workspace, sha).await.map_err(failed)?;
|
||||
return Ok(WorkspaceAction::Reset);
|
||||
}
|
||||
let key = targets
|
||||
.iter()
|
||||
.find(|(_, candidate)| candidate == sha)
|
||||
.map_or(targets[0].0.key, |(target, _)| target.key);
|
||||
workspaces
|
||||
.restore(workspace, key, sha)
|
||||
.await
|
||||
.map_err(failed)?;
|
||||
Ok(WorkspaceAction::Restored)
|
||||
}
|
||||
|
|
@ -1,5 +1,71 @@
|
|||
//! Petri's test kit, for Fabro crates that check a store implementation
|
||||
//! against Petri's contract from their own tests. Compiled only with the
|
||||
//! against Petri's contract from their own tests, and an in-memory platform
|
||||
//! record store for tests of the hooks and recovery. Compiled only with the
|
||||
//! `test-support` feature, which a dev-dependency turns on.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::sync::{Mutex, MutexGuard, PoisonError};
|
||||
|
||||
use async_trait::async_trait;
|
||||
use fabro_store::platform_records::now_ms;
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind, StagePosition, StoredPlatformRecord};
|
||||
use fabro_types::RunId;
|
||||
pub use petri_testkit::run_store;
|
||||
|
||||
use crate::platform_records::{PlatformRecordError, PlatformRecords};
|
||||
|
||||
/// Platform records kept in memory, per run, in seq order.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct MemoryPlatformRecords {
|
||||
runs: Mutex<HashMap<RunId, Vec<StoredPlatformRecord>>>,
|
||||
}
|
||||
|
||||
impl MemoryPlatformRecords {
|
||||
#[must_use]
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Every record of the run, in seq order.
|
||||
#[must_use]
|
||||
pub fn records(&self, run_id: &RunId) -> Vec<StoredPlatformRecord> {
|
||||
lock(&self.runs).get(run_id).cloned().unwrap_or_default()
|
||||
}
|
||||
}
|
||||
|
||||
fn lock<T>(mutex: &Mutex<T>) -> MutexGuard<'_, T> {
|
||||
mutex.lock().unwrap_or_else(PoisonError::into_inner)
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl PlatformRecords for MemoryPlatformRecords {
|
||||
async fn append(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
record: &PlatformRecord,
|
||||
position: Option<StagePosition>,
|
||||
) -> Result<StoredPlatformRecord, PlatformRecordError> {
|
||||
let mut runs = lock(&self.runs);
|
||||
let records = runs.entry(*run_id).or_default();
|
||||
let stored = StoredPlatformRecord {
|
||||
seq: records.len() as u64 + 1,
|
||||
recorded_at: now_ms(),
|
||||
record: record.clone(),
|
||||
position,
|
||||
};
|
||||
records.push(stored.clone());
|
||||
Ok(stored)
|
||||
}
|
||||
|
||||
async fn read_kind(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
kind: PlatformRecordKind,
|
||||
) -> Result<Vec<StoredPlatformRecord>, PlatformRecordError> {
|
||||
Ok(self
|
||||
.records(run_id)
|
||||
.into_iter()
|
||||
.filter(|record| record.record.kind() == kind)
|
||||
.collect())
|
||||
}
|
||||
}
|
||||
|
|
|
|||
116
lib/components/fabro-petri/src/workspace.rs
Normal file
116
lib/components/fabro-petri/src/workspace.rs
Normal file
|
|
@ -0,0 +1,116 @@
|
|||
//! Which workspace a Petri scope runs in, from the run's own records.
|
||||
//!
|
||||
//! Petri names an isolated scope's workspace after its invocation and scope
|
||||
//! (`invocation-<n>-scope-<m>`), and a nested invocation that inherits its
|
||||
//! caller's sandbox shares the caller's workspace through the lease the
|
||||
//! coordinator recorded. The hooks and recovery both need the workspace id
|
||||
//! behind a scope, and both read it the same way here: the direct name
|
||||
//! when its workspace exists on this host, else the lease the invocation's
|
||||
//! declaration names, resolved through the resource log.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use petri_execution::host::{self, HostError};
|
||||
use petri_execution::{
|
||||
Access, InvocationId, ResourceError, ResourceStore, RunKey, RunLogs, RunStore, SandboxBinding,
|
||||
};
|
||||
use petri_runtime::executor::WorkspaceId;
|
||||
use petri_runtime::ir::ScopeId;
|
||||
use petri_store::StoreError;
|
||||
|
||||
/// Why a workspace could not be named from the run's records.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum WorkspaceLookupError {
|
||||
#[error("the run's record could not be opened")]
|
||||
Open(#[source] StoreError),
|
||||
#[error("the run's coordinator state could not be read")]
|
||||
State(#[source] HostError),
|
||||
#[error("the run's resource log could not be read")]
|
||||
Resources(#[source] ResourceError),
|
||||
#[error("invocation {invocation} is not in the run's record")]
|
||||
UnknownInvocation { invocation: InvocationId },
|
||||
}
|
||||
|
||||
/// The workspace id of an isolated scope: what the coordinator allocates
|
||||
/// for `scope` in `invocation`.
|
||||
#[must_use]
|
||||
pub fn isolated_workspace(invocation: InvocationId, scope: ScopeId) -> String {
|
||||
WorkspaceId::scoped(Some(&invocation.workspace_prefix()), scope)
|
||||
.as_str()
|
||||
.to_owned()
|
||||
}
|
||||
|
||||
/// The workspaces of a run, read from its records through a handle that
|
||||
/// holds no lease.
|
||||
pub struct WorkspaceLookup {
|
||||
store: Arc<dyn RunStore>,
|
||||
key: RunKey,
|
||||
}
|
||||
|
||||
impl WorkspaceLookup {
|
||||
#[must_use]
|
||||
pub fn new(store: Arc<dyn RunStore>, key: RunKey) -> Self {
|
||||
Self { store, key }
|
||||
}
|
||||
|
||||
async fn logs(&self) -> Result<Arc<dyn RunLogs>, WorkspaceLookupError> {
|
||||
self.store
|
||||
.open(&self.key, Access::Read)
|
||||
.await
|
||||
.map_err(WorkspaceLookupError::Open)
|
||||
}
|
||||
|
||||
/// The workspace an invocation inherited from its caller, or `None`
|
||||
/// when the invocation owns its sandboxes.
|
||||
pub async fn inherited(
|
||||
&self,
|
||||
invocation: InvocationId,
|
||||
) -> Result<Option<String>, WorkspaceLookupError> {
|
||||
let logs = self.logs().await?;
|
||||
let state = host::stored_state(&*logs)
|
||||
.await
|
||||
.map_err(WorkspaceLookupError::State)?;
|
||||
let declared = state
|
||||
.invocations
|
||||
.get(&invocation)
|
||||
.ok_or(WorkspaceLookupError::UnknownInvocation { invocation })?;
|
||||
match declared.declaration.sandbox {
|
||||
SandboxBinding::Isolated => Ok(None),
|
||||
SandboxBinding::Inherited { lease } => {
|
||||
let resources = ResourceStore::load(&logs)
|
||||
.await
|
||||
.map_err(WorkspaceLookupError::Resources)?;
|
||||
let record = resources
|
||||
.resolve(lease)
|
||||
.map_err(WorkspaceLookupError::Resources)?;
|
||||
Ok(Some(record.workspace.as_str().to_owned()))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Every workspace an invocation runs in: its own leases' workspaces,
|
||||
/// or the one it inherited.
|
||||
pub async fn of_invocation(
|
||||
&self,
|
||||
invocation: InvocationId,
|
||||
) -> Result<Vec<String>, WorkspaceLookupError> {
|
||||
if let Some(inherited) = self.inherited(invocation).await? {
|
||||
return Ok(vec![inherited]);
|
||||
}
|
||||
let logs = self.logs().await?;
|
||||
let resources = ResourceStore::load(&logs)
|
||||
.await
|
||||
.map_err(WorkspaceLookupError::Resources)?;
|
||||
let mut workspaces: Vec<String> = resources
|
||||
.records()
|
||||
.filter(|record| {
|
||||
record.allocation.invocation == invocation
|
||||
&& record.state != petri_execution::LeaseState::Deleted
|
||||
})
|
||||
.map(|record| record.workspace.as_str().to_owned())
|
||||
.collect();
|
||||
workspaces.sort();
|
||||
workspaces.dedup();
|
||||
Ok(workspaces)
|
||||
}
|
||||
}
|
||||
691
lib/components/fabro-petri/tests/hooks.rs
Normal file
691
lib/components/fabro-petri/tests/hooks.rs
Normal file
|
|
@ -0,0 +1,691 @@
|
|||
//! Fabro's hooks on a Petri run, in process: command-only bundles run
|
||||
//! through `engine::run` on the host sandbox with the memory store and
|
||||
//! in-memory platform records, and the checkpoint commit, its record, the
|
||||
//! failure route, the fatal checkpoint, and the run-end hooks are checked
|
||||
//! against the workspace's Git history and the run's records.
|
||||
//!
|
||||
//! Every run acquires its scope through the sandbox-driver host plugin, so
|
||||
//! the tests skip when that executable is not found, unless
|
||||
//! `FABRO_REQUIRE_SANDBOX_PLUGINS` is set. The crash cases of the recovery
|
||||
//! protocol need a worker to kill and live in the CLI's scenario suite.
|
||||
|
||||
#![expect(
|
||||
clippy::disallowed_methods,
|
||||
reason = "the tests locate the plugin executable through the process environment and read the workspace's history with git"
|
||||
)]
|
||||
#![expect(clippy::print_stderr, reason = "a skipped test says why on its stderr")]
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
use std::env;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Arc;
|
||||
|
||||
use fabro_checkpoint::author::GitAuthor;
|
||||
use fabro_petri::admission::AdmittedGraphs;
|
||||
use fabro_petri::check::{self, Bundle, CheckRequest, Launch};
|
||||
use fabro_petri::checkpoint::{CHECKPOINT_FAILED_CLASS, CheckpointKey, RunWorkspaces};
|
||||
use fabro_petri::engine::{self, Execution, RunRequest, RunStatus};
|
||||
use fabro_petri::hooks::HooksSpec;
|
||||
use fabro_petri::platform_records::PlatformRecords;
|
||||
use fabro_petri::recovery::{self, Recovery, RecoveryRequest};
|
||||
use fabro_petri::runtime::RuntimeSpec;
|
||||
use fabro_petri::test_support::MemoryPlatformRecords;
|
||||
use fabro_store::{PlatformRecord, PlatformRecordKind};
|
||||
use fabro_types::settings::run::RunCheckpointSettings;
|
||||
use fabro_types::{RunId, SandboxProviderKind};
|
||||
use petri_execution::inspect::{self, RunInspection};
|
||||
use petri_store::{Access, MemoryRunStore, RunKey, RunStore as _};
|
||||
use tokio::fs;
|
||||
use tokio::process::Command;
|
||||
use tokio_util::sync::CancellationToken;
|
||||
|
||||
mod support;
|
||||
|
||||
const HOST_PLUGIN: &str = "sandbox-driver-host";
|
||||
const HOST_PLUGIN_OVERRIDE: &str = "PETRI_SANDBOX_HOST_PLUGIN";
|
||||
const REQUIRE_ENV: &str = "FABRO_REQUIRE_SANDBOX_PLUGINS";
|
||||
|
||||
/// The host plugin as Petri's lookup finds it: the override variable, else
|
||||
/// the executable on `PATH`. `None`, after saying so, when the test should
|
||||
/// skip; a panic when the environment forbids a skip.
|
||||
fn host_plugin() -> Option<PathBuf> {
|
||||
let found = env::var_os(HOST_PLUGIN_OVERRIDE)
|
||||
.map(PathBuf::from)
|
||||
.or_else(|| {
|
||||
env::split_paths(&env::var_os("PATH")?)
|
||||
.map(|dir| dir.join(HOST_PLUGIN))
|
||||
.find(|candidate| candidate.is_file())
|
||||
});
|
||||
if found.is_none() {
|
||||
assert!(
|
||||
env::var_os(REQUIRE_ENV).is_none(),
|
||||
"{REQUIRE_ENV} is set, but {HOST_PLUGIN} is not on PATH and {HOST_PLUGIN_OVERRIDE} is unset"
|
||||
);
|
||||
eprintln!("skipping: {HOST_PLUGIN} is not on PATH and {HOST_PLUGIN_OVERRIDE} is unset");
|
||||
}
|
||||
found
|
||||
}
|
||||
|
||||
/// A command-only bundle: the stage lines go between `start` and `exit`,
|
||||
/// the edge lines after them.
|
||||
fn workflow(stages: &str, edges: &str) -> String {
|
||||
format!(
|
||||
"digraph Hooks {{\n graph [goal=\"Check the hooks\", default_max_retries=0]\n start \
|
||||
[shape=Mdiamond]\n exit [shape=Msquare]\n{stages}\n{edges}\n}}\n"
|
||||
)
|
||||
}
|
||||
|
||||
const SETTINGS: &str = "_version = 1\n\n[workflow]\ngraph = \"workflow.fabro\"\n";
|
||||
|
||||
/// The bundle admitted the way the create handler admits it.
|
||||
fn admit(workflow: &str, settings: &str) -> AdmittedGraphs {
|
||||
let request = CheckRequest {
|
||||
bundle: Bundle {
|
||||
files: BTreeMap::from([
|
||||
("workflow.fabro".to_string(), workflow.to_string()),
|
||||
("workflow.toml".to_string(), settings.to_string()),
|
||||
]),
|
||||
entrypoint: "workflow.fabro".to_string(),
|
||||
project_toml: None,
|
||||
},
|
||||
inputs: BTreeMap::new(),
|
||||
launch: Launch::default(),
|
||||
runtime: RuntimeSpec::default(),
|
||||
};
|
||||
let admitted = check::check(&request).expect("the bundle is admitted");
|
||||
AdmittedGraphs {
|
||||
graph: admitted.graph,
|
||||
children: admitted.children,
|
||||
}
|
||||
}
|
||||
|
||||
/// One run's pieces: the store, its platform records, where it ran.
|
||||
struct Harness {
|
||||
run_id: RunId,
|
||||
run_dir: PathBuf,
|
||||
store: Arc<MemoryRunStore>,
|
||||
records: Arc<MemoryPlatformRecords>,
|
||||
_root: tempfile::TempDir,
|
||||
}
|
||||
|
||||
impl Harness {
|
||||
fn new() -> Self {
|
||||
let root = tempfile::tempdir().expect("a temp dir");
|
||||
Self {
|
||||
run_id: RunId::new(),
|
||||
run_dir: root.path().join("run"),
|
||||
store: Arc::new(MemoryRunStore::new()),
|
||||
records: Arc::new(MemoryPlatformRecords::new()),
|
||||
_root: root,
|
||||
}
|
||||
}
|
||||
|
||||
fn hooks(&self) -> HooksSpec {
|
||||
HooksSpec {
|
||||
records: Arc::clone(&self.records) as Arc<dyn PlatformRecords>,
|
||||
author: GitAuthor::default(),
|
||||
checkpoint: RunCheckpointSettings::default(),
|
||||
host_workspaces: true,
|
||||
test_gates: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Run the bundle to its end through the engine module, as the worker
|
||||
/// does, and report what the record says.
|
||||
async fn run(&self, workflow: &str, settings: &str) -> engine::RunOutcome {
|
||||
let (interviewer, observers) = no_questions();
|
||||
let request = RunRequest {
|
||||
run_id: self.run_id.to_string(),
|
||||
run_dir: self.run_dir.clone(),
|
||||
execution: Execution::Start(admit(workflow, settings)),
|
||||
store: Arc::clone(&self.store) as Arc<dyn petri_store::RunStore>,
|
||||
runtime: RuntimeSpec::default(),
|
||||
provider: SandboxProviderKind::LOCAL,
|
||||
cancel: CancellationToken::new(),
|
||||
interviewer,
|
||||
observers,
|
||||
secrets: None,
|
||||
blobs: None,
|
||||
hooks: Some(self.hooks()),
|
||||
};
|
||||
engine::run(request).await.expect("the run executes")
|
||||
}
|
||||
|
||||
async fn inspection(&self) -> RunInspection {
|
||||
let logs = self
|
||||
.store
|
||||
.open(&RunKey::new(self.run_id.to_string()), Access::Read)
|
||||
.await
|
||||
.expect("the run opens for reading");
|
||||
inspect::inspect_run(&*logs)
|
||||
.await
|
||||
.expect("the stored run inspects")
|
||||
}
|
||||
|
||||
fn workspaces(&self) -> RunWorkspaces {
|
||||
RunWorkspaces::new(
|
||||
self.run_dir.clone(),
|
||||
self.run_id.to_string(),
|
||||
GitAuthor::default(),
|
||||
&RunCheckpointSettings::default(),
|
||||
)
|
||||
}
|
||||
|
||||
/// The one workspace the run's root scope used.
|
||||
async fn workspace(&self) -> String {
|
||||
let mut entries = fs::read_dir(self.run_dir.join("scopes"))
|
||||
.await
|
||||
.expect("the scopes directory exists");
|
||||
let mut names = Vec::new();
|
||||
while let Some(entry) = entries.next_entry().await.expect("an entry reads") {
|
||||
names.push(entry.file_name().to_string_lossy().into_owned());
|
||||
}
|
||||
assert_eq!(names.len(), 1, "one workspace: {names:?}");
|
||||
names.remove(0)
|
||||
}
|
||||
|
||||
fn workspace_path(&self, workspace: &str) -> PathBuf {
|
||||
self.workspaces().workspace_path(workspace)
|
||||
}
|
||||
|
||||
/// The checkpoint records, in seq order, as `(key, sha)`.
|
||||
fn checkpoints(&self) -> Vec<(CheckpointKey, String)> {
|
||||
self.records
|
||||
.records(&self.run_id)
|
||||
.into_iter()
|
||||
.filter_map(|stored| match stored.record {
|
||||
PlatformRecord::Checkpoint(record) => Some((
|
||||
CheckpointKey::from_operation(record.operation.as_ref()?)?,
|
||||
record.git_commit_sha?,
|
||||
)),
|
||||
_ => None,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
async fn recover(&self) -> Recovery {
|
||||
recovery::recover(RecoveryRequest {
|
||||
run_id: self.run_id,
|
||||
run_dir: self.run_dir.clone(),
|
||||
store: Arc::clone(&self.store) as Arc<dyn petri_store::RunStore>,
|
||||
records: Arc::clone(&self.records) as Arc<dyn PlatformRecords>,
|
||||
author: GitAuthor::default(),
|
||||
checkpoint: RunCheckpointSettings::default(),
|
||||
host_workspaces: true,
|
||||
})
|
||||
.await
|
||||
.expect("recovery decides")
|
||||
}
|
||||
}
|
||||
|
||||
/// An interviewer for a run that asks nothing, with its expiry observer.
|
||||
fn no_questions() -> (
|
||||
Arc<dyn petri_execution::Interviewer>,
|
||||
Vec<Arc<dyn petri_execution::ExecutionObserver>>,
|
||||
) {
|
||||
let interviewer = support::no_questions(Arc::new(support::Silent));
|
||||
let observers = vec![interviewer.observer()];
|
||||
(Arc::new(interviewer), observers)
|
||||
}
|
||||
|
||||
/// `git` in a workspace, its stdout.
|
||||
async fn git(path: &Path, args: &[&str]) -> String {
|
||||
let output = Command::new("git")
|
||||
.args(args)
|
||||
.current_dir(path)
|
||||
.output()
|
||||
.await
|
||||
.expect("git runs");
|
||||
assert!(
|
||||
output.status.success(),
|
||||
"git {args:?} failed: {}",
|
||||
String::from_utf8_lossy(&output.stderr)
|
||||
);
|
||||
String::from_utf8_lossy(&output.stdout).trim().to_string()
|
||||
}
|
||||
|
||||
/// The commits on the run branch, oldest first, as `(sha, subject, key)`.
|
||||
async fn commits(path: &Path) -> Vec<(String, String, Option<CheckpointKey>)> {
|
||||
let log = git(path, &["log", "--reverse", "--format=%H%x00%s%x00%B%x1e"]).await;
|
||||
log.split('\u{1e}')
|
||||
.filter(|entry| !entry.trim().is_empty())
|
||||
.map(|entry| {
|
||||
let mut parts = entry.trim_start().splitn(3, '\0');
|
||||
let sha = parts.next().unwrap_or_default().to_string();
|
||||
let subject = parts.next().unwrap_or_default().to_string();
|
||||
let body = parts.next().unwrap_or_default();
|
||||
(sha, subject, CheckpointKey::from_message(body))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The stages that ran, by node name, with their final status.
|
||||
fn stages(inspection: &RunInspection) -> Vec<(String, String)> {
|
||||
inspection
|
||||
.executions
|
||||
.iter()
|
||||
.filter_map(|execution| execution.engine.as_ref())
|
||||
.flat_map(|engine| engine.history.iter())
|
||||
.map(|record| (record.node.to_string(), record.status.to_string()))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Every finished stage is committed on the run branch with the identity
|
||||
/// trailers, its platform record names the commit, and the run-end hooks
|
||||
/// reached Petri's local service through Fabro's wrapper.
|
||||
#[tokio::test]
|
||||
async fn every_finish_is_committed_and_recorded() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
let workflow = workflow(
|
||||
" write [shape=parallelogram, script=\"echo one > out.txt\"]\n check \
|
||||
[shape=parallelogram, script=\"test \\\"$(cat out.txt)\\\" = one\"]",
|
||||
" start -> write -> check -> exit",
|
||||
);
|
||||
let settings = format!(
|
||||
"{SETTINGS}\n[[run.hooks]]\nevent = \"run_complete\"\nscript = \"echo run_complete >> \
|
||||
run-end.log\"\n\n[[run.hooks]]\nevent = \"sandbox_cleanup\"\nscript = \"echo \
|
||||
sandbox_cleanup >> run-end.log\"\n"
|
||||
);
|
||||
let outcome = harness.run(&workflow, &settings).await;
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
assert!(outcome.complete, "{:?}", outcome.incomplete);
|
||||
|
||||
let workspace = harness.workspace().await;
|
||||
let path = harness.workspace_path(&workspace);
|
||||
assert_eq!(
|
||||
git(&path, &["rev-parse", "--abbrev-ref", "HEAD"]).await,
|
||||
format!("fabro/run/{}", harness.run_id)
|
||||
);
|
||||
let commits = commits(&path).await;
|
||||
let subjects: Vec<&str> = commits
|
||||
.iter()
|
||||
.map(|(_, subject, _)| subject.as_str())
|
||||
.collect();
|
||||
let run_id = harness.run_id.to_string();
|
||||
assert_eq!(subjects, vec![
|
||||
format!("fabro({run_id}): start (success)"),
|
||||
format!("fabro({run_id}): write (success)"),
|
||||
format!("fabro({run_id}): check (success)"),
|
||||
format!("fabro({run_id}): exit (success)"),
|
||||
]);
|
||||
assert!(
|
||||
commits.iter().all(|(_, _, key)| key.is_some()),
|
||||
"every commit carries its key: {commits:?}"
|
||||
);
|
||||
|
||||
let checkpoints = harness.checkpoints();
|
||||
assert_eq!(checkpoints.len(), 4, "{checkpoints:?}");
|
||||
let by_sha: Vec<&String> = checkpoints.iter().map(|(_, sha)| sha).collect();
|
||||
let committed: Vec<&String> = commits.iter().map(|(sha, _, _)| sha).collect();
|
||||
assert_eq!(by_sha, committed, "each record names its stage's commit");
|
||||
for ((key, _), (_, _, trailer)) in checkpoints.iter().zip(&commits) {
|
||||
assert_eq!(Some(*key), *trailer);
|
||||
}
|
||||
assert!(
|
||||
checkpoints.iter().all(|(key, _)| key.execution == 0),
|
||||
"{checkpoints:?}"
|
||||
);
|
||||
|
||||
// The snapshot repository holds every checkpoint.
|
||||
let published = harness
|
||||
.workspaces()
|
||||
.published(&workspace)
|
||||
.await
|
||||
.expect("the snapshots list");
|
||||
assert_eq!(published.len(), 4);
|
||||
|
||||
// `run_complete` and `sandbox_cleanup` ran through the forwarded
|
||||
// service, with the sandbox in place.
|
||||
let run_end = fs::read_to_string(path.join("run-end.log"))
|
||||
.await
|
||||
.expect("the run-end hooks wrote their log");
|
||||
assert_eq!(run_end, "run_complete\nsandbox_cleanup\n");
|
||||
}
|
||||
|
||||
/// A stage that fails on its own terms is committed like a successful one,
|
||||
/// and its failure route runs on the committed files.
|
||||
#[tokio::test]
|
||||
async fn a_failed_stage_is_committed_and_its_route_sees_the_files() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
let workflow = workflow(
|
||||
" work [shape=parallelogram, script=\"echo partial > out.txt; exit 1\"]\n fix \
|
||||
[shape=parallelogram, script=\"test \\\"$(cat out.txt)\\\" = partial && echo fixed >> \
|
||||
out.txt\"]",
|
||||
" start -> work -> exit\n work -> fix [condition=\"outcome=failed\"]\n fix -> exit",
|
||||
);
|
||||
let outcome = harness.run(&workflow, SETTINGS).await;
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
let inspection = harness.inspection().await;
|
||||
assert_eq!(stages(&inspection), vec![
|
||||
("start".to_string(), "success".to_string()),
|
||||
("work".to_string(), "failure".to_string()),
|
||||
("fix".to_string(), "success".to_string()),
|
||||
("exit".to_string(), "success".to_string()),
|
||||
]);
|
||||
|
||||
let workspace = harness.workspace().await;
|
||||
let path = harness.workspace_path(&workspace);
|
||||
let commits = commits(&path).await;
|
||||
let run_id = harness.run_id.to_string();
|
||||
assert_eq!(commits[1].1, format!("fabro({run_id}): work (failure)"));
|
||||
assert_eq!(
|
||||
git(&path, &["show", &format!("{}:out.txt", commits[1].0)]).await,
|
||||
"partial",
|
||||
"the failed stage's files are in its snapshot"
|
||||
);
|
||||
assert_eq!(
|
||||
git(&path, &["show", &format!("{}:out.txt", commits[2].0)]).await,
|
||||
"partial\nfixed",
|
||||
"the route ran on the committed files"
|
||||
);
|
||||
assert_eq!(harness.checkpoints().len(), 4);
|
||||
}
|
||||
|
||||
/// A checkpoint commit that fails is fatal: the stage's outcome is recorded
|
||||
/// as `checkpoint_failed`, no route is taken, the run ends failed with the
|
||||
/// checkpoint's error, and a restart reports it failed without resuming.
|
||||
#[tokio::test]
|
||||
async fn a_failed_checkpoint_ends_the_run_with_no_route() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
let workflow = workflow(
|
||||
" wreck [shape=parallelogram, script=\"rm -rf .git && echo garbage > .git && echo wrecked \
|
||||
> out.txt\"]\n next [shape=parallelogram, script=\"echo next > next.txt\"]\n fix \
|
||||
[shape=parallelogram, script=\"echo fix > fix.txt\"]",
|
||||
" start -> wreck -> next -> exit\n wreck -> fix [condition=\"outcome=failed\"]\n fix -> \
|
||||
exit",
|
||||
);
|
||||
let outcome = harness.run(&workflow, SETTINGS).await;
|
||||
assert_eq!(outcome.status, RunStatus::Failed, "{outcome:?}");
|
||||
let failure = outcome
|
||||
.failure
|
||||
.clone()
|
||||
.expect("the run failed with a reason");
|
||||
assert!(
|
||||
failure.contains("checkpoint commit of `wreck` failed"),
|
||||
"{failure}"
|
||||
);
|
||||
|
||||
let inspection = harness.inspection().await;
|
||||
let attempts: Vec<_> = inspection
|
||||
.executions
|
||||
.iter()
|
||||
.filter_map(|execution| execution.engine.as_ref())
|
||||
.flat_map(|engine| engine.attempts.iter())
|
||||
.collect();
|
||||
let wreck = attempts
|
||||
.iter()
|
||||
.find(|attempt| attempt.node.as_deref() == Some("wreck"))
|
||||
.expect("the wrecked stage finished");
|
||||
assert_eq!(wreck.status, "failure");
|
||||
assert_eq!(
|
||||
wreck.failure.as_ref().map(|failure| failure.class.as_str()),
|
||||
Some(CHECKPOINT_FAILED_CLASS),
|
||||
"{wreck:?}"
|
||||
);
|
||||
assert!(
|
||||
attempts
|
||||
.iter()
|
||||
.all(|attempt| !matches!(attempt.node.as_deref(), Some("next" | "fix"))),
|
||||
"no route ran: {:?}",
|
||||
stages(&inspection)
|
||||
);
|
||||
let workspace = harness.workspace().await;
|
||||
let path = harness.workspace_path(&workspace);
|
||||
assert!(!fs::try_exists(path.join("next.txt")).await.expect("exists"));
|
||||
assert!(!fs::try_exists(path.join("fix.txt")).await.expect("exists"));
|
||||
// The wrecked stage has no record: its commit never landed.
|
||||
let checkpoints = harness.checkpoints();
|
||||
assert_eq!(checkpoints.len(), 1, "{checkpoints:?}");
|
||||
|
||||
// A restart finds the failed checkpoint and reports the run failed.
|
||||
let recovery = harness.recover().await;
|
||||
assert!(
|
||||
matches!(&recovery, Recovery::Failed { reason } if reason.contains("checkpoint of wreck")),
|
||||
"{recovery:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The two ends of recovery that need no crash: a run the store never held
|
||||
/// starts over, and a run that finished has nothing to bring back.
|
||||
#[tokio::test]
|
||||
async fn recovery_starts_an_unknown_run_and_resumes_a_finished_one() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
assert_eq!(harness.recover().await, Recovery::Start);
|
||||
|
||||
let workflow = workflow(
|
||||
" write [shape=parallelogram, script=\"echo one > out.txt\"]",
|
||||
" start -> write -> exit",
|
||||
);
|
||||
let outcome = harness.run(&workflow, SETTINGS).await;
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
assert_eq!(harness.recover().await, Recovery::Resume {
|
||||
workspaces: Vec::new(),
|
||||
});
|
||||
assert_eq!(
|
||||
harness
|
||||
.records
|
||||
.read_kind(&harness.run_id, PlatformRecordKind::Checkpoint)
|
||||
.await
|
||||
.expect("the records read")
|
||||
.len(),
|
||||
3
|
||||
);
|
||||
}
|
||||
|
||||
/// A `[[run.hooks]]` hook that blocks a tool effect keeps working through
|
||||
/// the forwarded local service: the agent's `rm` is refused by the
|
||||
/// `pre_tool_use` hook, the model is told why, and the file it aimed at is
|
||||
/// still in the stage's snapshot.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn a_run_hook_blocks_a_tool_effect_through_the_forwarded_service() {
|
||||
use fabro_auth::test_support::env_credential_source;
|
||||
use fabro_llm::test_support::test_catalog_with_provider_base_url;
|
||||
use fabro_petri::runtime;
|
||||
use fabro_test::{TwinScenario, TwinScenarios, TwinToolCall};
|
||||
use lithos_llm::catalog::ProviderId;
|
||||
use serde_json::json;
|
||||
|
||||
const MODEL: &str = "gpt-5.6-sol";
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let twin = fabro_test::twin_openai().await;
|
||||
let namespace = format!("{}::{}", module_path!(), line!());
|
||||
TwinScenarios::new(namespace.clone())
|
||||
.scenario(
|
||||
TwinScenario::responses(MODEL)
|
||||
.input_contains("Remove the scratch file")
|
||||
.tool_call(TwinToolCall::new(
|
||||
"shell_command",
|
||||
json!({ "command": "rm -f scratch.txt && echo removed" }),
|
||||
)),
|
||||
)
|
||||
.scenario(
|
||||
TwinScenario::responses(MODEL)
|
||||
.input_contains("destructive commands are not allowed")
|
||||
.text("Understood, the file stays."),
|
||||
)
|
||||
.load(twin)
|
||||
.await;
|
||||
let api_key = namespace.clone();
|
||||
let credentials =
|
||||
env_credential_source(move |name| (name == "OPENAI_API_KEY").then(|| api_key.clone()));
|
||||
let client = runtime::model_client(
|
||||
test_catalog_with_provider_base_url("openai", &twin.base_url),
|
||||
credentials,
|
||||
None,
|
||||
&[ProviderId::new("openai")],
|
||||
)
|
||||
.expect("the model client builds")
|
||||
.expect("openai is eligible");
|
||||
|
||||
let harness = Harness::new();
|
||||
let (interviewer, observers) = no_questions();
|
||||
let workflow = format!(
|
||||
"digraph Hooks {{\n graph [backend=\"api\", goal=\"Check the tool hooks\", \
|
||||
default_max_retries=0]\n start [shape=Mdiamond]\n exit [shape=Msquare]\n seed \
|
||||
[shape=parallelogram, script=\"echo keep > scratch.txt\"]\n agent [prompt=\"Remove the \
|
||||
scratch file with the shell tool.\", model=\"{MODEL}\", provider=\"openai\", \
|
||||
fidelity=\"full\"]\n check [shape=parallelogram, script=\"test \\\"$(cat scratch.txt)\\\" \
|
||||
= keep\"]\n start -> seed -> agent -> check -> exit\n}}\n"
|
||||
);
|
||||
let settings = format!(
|
||||
"{SETTINGS}\n[[run.hooks]]\nname = \"no-destruction\"\nevent = \"pre_tool_use\"\nmatcher \
|
||||
= \"shell\"\nscript = \"if grep -q 'rm ' \\\"$FABRO_HOOK_CONTEXT\\\"; then echo \
|
||||
'{{\\\"decision\\\":\\\"block\\\",\\\"reason\\\":\\\"destructive commands are not \
|
||||
allowed\\\"}}'; exit 2; fi\"\n"
|
||||
);
|
||||
let request = RunRequest {
|
||||
run_id: harness.run_id.to_string(),
|
||||
run_dir: harness.run_dir.clone(),
|
||||
execution: Execution::Start(admit(&workflow, &settings)),
|
||||
store: Arc::clone(&harness.store) as Arc<dyn petri_store::RunStore>,
|
||||
runtime: RuntimeSpec {
|
||||
model_client: Some(client),
|
||||
..RuntimeSpec::default()
|
||||
},
|
||||
provider: SandboxProviderKind::LOCAL,
|
||||
cancel: CancellationToken::new(),
|
||||
interviewer,
|
||||
observers,
|
||||
secrets: None,
|
||||
blobs: None,
|
||||
hooks: Some(harness.hooks()),
|
||||
};
|
||||
let outcome = engine::run(request).await.expect("the run executes");
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
let inspection = harness.inspection().await;
|
||||
assert_eq!(stages(&inspection), vec![
|
||||
("start".to_string(), "success".to_string()),
|
||||
("seed".to_string(), "success".to_string()),
|
||||
("agent".to_string(), "success".to_string()),
|
||||
("check".to_string(), "success".to_string()),
|
||||
("exit".to_string(), "success".to_string()),
|
||||
]);
|
||||
let requests = twin.request_logs(&namespace).await;
|
||||
let inputs: Vec<&str> = requests["requests"]
|
||||
.as_array()
|
||||
.map(|requests| {
|
||||
requests
|
||||
.iter()
|
||||
.filter_map(|request| request["input_text"].as_str())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
assert_eq!(inputs.len(), 2, "{inputs:?}");
|
||||
assert!(
|
||||
inputs[1].contains("destructive commands are not allowed"),
|
||||
"the model was told why the tool was blocked: {inputs:?}"
|
||||
);
|
||||
let workspace = harness.workspace().await;
|
||||
let path = harness.workspace_path(&workspace);
|
||||
assert_eq!(
|
||||
git(&path, &["show", "HEAD:scratch.txt"]).await,
|
||||
"keep",
|
||||
"the blocked removal never happened"
|
||||
);
|
||||
assert_eq!(harness.checkpoints().len(), 5);
|
||||
}
|
||||
|
||||
/// The run's records name the workspace its root invocation ran in: what
|
||||
/// recovery reads to find the workspace of a live execution.
|
||||
#[tokio::test]
|
||||
async fn the_records_name_the_root_invocations_workspace() {
|
||||
use fabro_petri::workspace::WorkspaceLookup;
|
||||
use petri_execution::InvocationId;
|
||||
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
let workflow = workflow(
|
||||
" write [shape=parallelogram, script=\"echo one > out.txt\"]",
|
||||
" start -> write -> exit",
|
||||
);
|
||||
let outcome = harness.run(&workflow, SETTINGS).await;
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
let lookup = WorkspaceLookup::new(
|
||||
Arc::clone(&harness.store) as Arc<dyn petri_store::RunStore>,
|
||||
RunKey::new(harness.run_id.to_string()),
|
||||
);
|
||||
let named = lookup
|
||||
.of_invocation(InvocationId::ROOT)
|
||||
.await
|
||||
.expect("the lookup reads the records");
|
||||
assert_eq!(named, vec![harness.workspace().await]);
|
||||
let inspection = harness.inspection().await;
|
||||
let statuses: Vec<&str> = inspection
|
||||
.executions
|
||||
.iter()
|
||||
.map(|execution| execution.status)
|
||||
.collect();
|
||||
assert_eq!(statuses, vec!["finished"]);
|
||||
}
|
||||
|
||||
/// The branches of a parallel node share their caller's workspace: their
|
||||
/// checkpoints are committed one at a time on the one run branch, every
|
||||
/// firing of every execution gets its record, and no transition reports a
|
||||
/// problem.
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn parallel_branches_checkpoint_the_shared_workspace_in_turn() {
|
||||
if host_plugin().is_none() {
|
||||
return;
|
||||
}
|
||||
let harness = Harness::new();
|
||||
let workflow = workflow(
|
||||
" fork [shape=component]\n a [shape=parallelogram, script=\"echo a > a.txt\"]\n b \
|
||||
[shape=parallelogram, script=\"echo b > b.txt\"]\n merge [shape=tripleoctagon]\n check \
|
||||
[shape=parallelogram, script=\"test -f a.txt && test -f b.txt\"]",
|
||||
" start -> fork\n fork -> a\n fork -> b\n a -> merge\n b -> merge\n merge -> check -> \
|
||||
exit",
|
||||
);
|
||||
let outcome = harness.run(&workflow, SETTINGS).await;
|
||||
assert_eq!(outcome.status, RunStatus::Success, "{outcome:?}");
|
||||
|
||||
let records = support::all_records(harness.store.as_ref(), &harness.run_id.to_string()).await;
|
||||
let problems: Vec<&serde_json::Value> = records
|
||||
.iter()
|
||||
.filter_map(|record| record.pointer("/event/$note"))
|
||||
.filter(|note| note["kind"] == "transition" && !note["payload"]["problems"].is_null())
|
||||
.collect();
|
||||
assert!(problems.is_empty(), "transition problems: {problems:#?}");
|
||||
|
||||
let inspection = harness.inspection().await;
|
||||
let mut finished: Vec<CheckpointKey> = inspection
|
||||
.executions
|
||||
.iter()
|
||||
.filter_map(|execution| Some((execution.execution.raw(), execution.engine.as_ref()?)))
|
||||
.flat_map(|(execution, engine)| {
|
||||
engine.attempts.iter().map(move |attempt| CheckpointKey {
|
||||
execution,
|
||||
firing: attempt.firing,
|
||||
attempt: attempt.attempt,
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
finished.sort();
|
||||
let mut recorded: Vec<CheckpointKey> = harness
|
||||
.checkpoints()
|
||||
.into_iter()
|
||||
.map(|(key, _)| key)
|
||||
.collect();
|
||||
recorded.sort();
|
||||
assert_eq!(
|
||||
recorded, finished,
|
||||
"every finish of every execution is recorded"
|
||||
);
|
||||
assert_eq!(inspection.executions.len(), 3, "the root and two branches");
|
||||
assert_eq!(harness.workspace().await, "invocation-0-scope-0");
|
||||
}
|
||||
|
|
@ -117,6 +117,7 @@ pub(crate) fn run_request(
|
|||
interviewer: Arc::new(interviewer),
|
||||
secrets: None,
|
||||
blobs: None,
|
||||
hooks: None,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -376,6 +376,13 @@ pub struct GitIdentityRecord {
|
|||
pub struct CheckpointRecord {
|
||||
pub execution: u64,
|
||||
pub firing: u64,
|
||||
/// The attempt whose files the commit holds; absent on a record written
|
||||
/// before the field existed.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub attempt: Option<u32>,
|
||||
/// The Petri workspace id the commit was made in.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub workspace: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub git_commit_sha: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
|
|
@ -857,6 +864,8 @@ mod tests {
|
|||
PlatformRecordKind::Checkpoint => PlatformRecord::Checkpoint(CheckpointRecord {
|
||||
execution: 0,
|
||||
firing: 3,
|
||||
attempt: Some(1),
|
||||
workspace: Some("invocation-0-scope-0".to_string()),
|
||||
git_commit_sha: Some("def".to_string()),
|
||||
diff_summary: Some(DiffSummary {
|
||||
files_changed: 1,
|
||||
|
|
|
|||
|
|
@ -261,7 +261,8 @@ impl RunSummaryStore {
|
|||
.unwrap_or_else(PoisonError::into_inner) = Some(hook);
|
||||
}
|
||||
|
||||
pub(crate) fn notify_platform_record(&self, run_id: RunId) {
|
||||
/// Wake the run's projector: a platform record was committed for the run.
|
||||
pub fn notify_platform_record(&self, run_id: RunId) {
|
||||
let hook = self
|
||||
.platform_hook
|
||||
.read()
|
||||
|
|
|
|||
|
|
@ -2039,6 +2039,45 @@ impl Client {
|
|||
Ok(response.into_inner().hash)
|
||||
}
|
||||
|
||||
/// Store one of Fabro's platform records for the run, tied to a Petri
|
||||
/// stage when it belongs to one, and get it back as stored.
|
||||
pub async fn append_petri_platform_record(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
body: types::PetriPlatformRecordAppendRequest,
|
||||
) -> Result<types::PetriPlatformRecord> {
|
||||
let response = self
|
||||
.send_api(|client| async move {
|
||||
client
|
||||
.append_petri_platform_record()
|
||||
.id(run_id.to_string())
|
||||
.body(body.clone())
|
||||
.send()
|
||||
.await
|
||||
})
|
||||
.await?;
|
||||
Ok(response.into_inner())
|
||||
}
|
||||
|
||||
/// The run's platform records in `seq` order, of one kind when `kind`
|
||||
/// names it.
|
||||
pub async fn list_petri_platform_records(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
kind: Option<&str>,
|
||||
) -> Result<Vec<types::PetriPlatformRecord>> {
|
||||
let response = self
|
||||
.send_api(|client| async move {
|
||||
let mut request = client.list_petri_platform_records().id(run_id.to_string());
|
||||
if let Some(kind) = kind {
|
||||
request = request.kind(kind);
|
||||
}
|
||||
request.send().await
|
||||
})
|
||||
.await?;
|
||||
Ok(response.into_inner().records)
|
||||
}
|
||||
|
||||
/// The blob with this digest, or `None` when the store holds no such
|
||||
/// blob. A run the store does not hold is an error.
|
||||
pub async fn read_petri_blob(
|
||||
|
|
|
|||
|
|
@ -38,6 +38,10 @@ impl EnvVars {
|
|||
pub const FABRO_TEST_IN_MEMORY_STORE: &'static str = "FABRO_TEST_IN_MEMORY_STORE";
|
||||
pub const FABRO_TEST_DISABLE_SPA_ASSETS: &'static str = "FABRO_TEST_DISABLE_SPA_ASSETS";
|
||||
pub const FABRO_TEST_MODE: &'static str = "FABRO_TEST_MODE";
|
||||
/// A directory of hold and release files a test uses to pause a Petri
|
||||
/// run's checkpoint at a named point (`fabro_petri::hooks`); unset
|
||||
/// outside tests.
|
||||
pub const FABRO_TEST_CHECKPOINT_GATES: &'static str = "FABRO_TEST_CHECKPOINT_GATES";
|
||||
pub const FABRO_VERBOSE: &'static str = "FABRO_VERBOSE";
|
||||
pub const FABRO_WEB_URL: &'static str = "FABRO_WEB_URL";
|
||||
pub const FABRO_WORKER_TOKEN: &'static str = "FABRO_WORKER_TOKEN";
|
||||
|
|
@ -222,6 +226,7 @@ mod tests {
|
|||
EnvVars::FABRO_TEST_IN_MEMORY_STORE,
|
||||
EnvVars::FABRO_TEST_DISABLE_SPA_ASSETS,
|
||||
EnvVars::FABRO_TEST_MODE,
|
||||
EnvVars::FABRO_TEST_CHECKPOINT_GATES,
|
||||
EnvVars::FABRO_VERBOSE,
|
||||
EnvVars::FABRO_WEB_URL,
|
||||
EnvVars::FABRO_WORKER_TOKEN,
|
||||
|
|
|
|||
|
|
@ -299,6 +299,9 @@ models/petri-access.ts
|
|||
models/petri-append-request.ts
|
||||
models/petri-open-request.ts
|
||||
models/petri-open-response.ts
|
||||
models/petri-platform-record-append-request.ts
|
||||
models/petri-platform-record-list.ts
|
||||
models/petri-platform-record.ts
|
||||
models/petri-record-list.ts
|
||||
models/petri-record.ts
|
||||
models/petri-release-request.ts
|
||||
|
|
|
|||
|
|
@ -40,6 +40,12 @@ import type { PetriOpenRequest } from '../models';
|
|||
// @ts-ignore
|
||||
import type { PetriOpenResponse } from '../models';
|
||||
// @ts-ignore
|
||||
import type { PetriPlatformRecord } from '../models';
|
||||
// @ts-ignore
|
||||
import type { PetriPlatformRecordAppendRequest } from '../models';
|
||||
// @ts-ignore
|
||||
import type { PetriPlatformRecordList } from '../models';
|
||||
// @ts-ignore
|
||||
import type { PetriRecordList } from '../models';
|
||||
// @ts-ignore
|
||||
import type { PetriReleaseRequest } from '../models';
|
||||
|
|
@ -66,6 +72,51 @@ import type { WriteRunBlobRequest } from '../models';
|
|||
*/
|
||||
export const RunInternalsApiAxiosParamCreator = function (configuration?: Configuration) {
|
||||
return {
|
||||
/**
|
||||
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
|
||||
* @summary Append Petri Platform Record
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
appendPetriPlatformRecord: async (id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options: RawAxiosRequestConfig = {}): Promise<RequestArgs> => {
|
||||
// verify required parameter 'id' is not null or undefined
|
||||
assertParamExists('appendPetriPlatformRecord', 'id', id)
|
||||
// verify required parameter 'petriPlatformRecordAppendRequest' is not null or undefined
|
||||
assertParamExists('appendPetriPlatformRecord', 'petriPlatformRecordAppendRequest', petriPlatformRecordAppendRequest)
|
||||
const localVarPath = `/api/v1/runs/{id}/petri/platform-records`
|
||||
.replace(`{${"id"}}`, encodeURIComponent(String(id)));
|
||||
// use dummy base URL string because the URL constructor only accepts absolute URLs.
|
||||
const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL);
|
||||
let baseOptions;
|
||||
if (configuration) {
|
||||
baseOptions = configuration.baseOptions;
|
||||
}
|
||||
|
||||
const localVarRequestOptions = { method: 'POST', ...baseOptions, ...options};
|
||||
const localVarHeaderParameter = {} as any;
|
||||
const localVarQueryParameter = {} as any;
|
||||
|
||||
// authentication SessionCookie required
|
||||
|
||||
// authentication BearerAuth required
|
||||
// http bearer authentication required
|
||||
await setBearerAuthToObject(localVarHeaderParameter, configuration)
|
||||
|
||||
localVarHeaderParameter['Content-Type'] = 'application/json';
|
||||
localVarHeaderParameter['Accept'] = 'application/json';
|
||||
|
||||
setSearchParams(localVarUrlObj, localVarQueryParameter);
|
||||
let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {};
|
||||
localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers};
|
||||
localVarRequestOptions.data = serializeDataIfNeeded(petriPlatformRecordAppendRequest, localVarRequestOptions, configuration)
|
||||
|
||||
return {
|
||||
url: toPathString(localVarUrlObj),
|
||||
options: localVarRequestOptions,
|
||||
};
|
||||
},
|
||||
/**
|
||||
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
|
||||
* @summary Append Petri Records
|
||||
|
|
@ -530,6 +581,51 @@ export const RunInternalsApiAxiosParamCreator = function (configuration?: Config
|
|||
options: localVarRequestOptions,
|
||||
};
|
||||
},
|
||||
/**
|
||||
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
|
||||
* @summary List Petri Platform Records
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {string} [kind] Only the platform records of this kind, as its `kind` tag spells it (`checkpoint`, `pull_request.created`, ...).
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
listPetriPlatformRecords: async (id: string, kind?: string, options: RawAxiosRequestConfig = {}): Promise<RequestArgs> => {
|
||||
// verify required parameter 'id' is not null or undefined
|
||||
assertParamExists('listPetriPlatformRecords', 'id', id)
|
||||
const localVarPath = `/api/v1/runs/{id}/petri/platform-records`
|
||||
.replace(`{${"id"}}`, encodeURIComponent(String(id)));
|
||||
// use dummy base URL string because the URL constructor only accepts absolute URLs.
|
||||
const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL);
|
||||
let baseOptions;
|
||||
if (configuration) {
|
||||
baseOptions = configuration.baseOptions;
|
||||
}
|
||||
|
||||
const localVarRequestOptions = { method: 'GET', ...baseOptions, ...options};
|
||||
const localVarHeaderParameter = {} as any;
|
||||
const localVarQueryParameter = {} as any;
|
||||
|
||||
// authentication SessionCookie required
|
||||
|
||||
// authentication BearerAuth required
|
||||
// http bearer authentication required
|
||||
await setBearerAuthToObject(localVarHeaderParameter, configuration)
|
||||
|
||||
if (kind !== undefined) {
|
||||
localVarQueryParameter['kind'] = kind;
|
||||
}
|
||||
|
||||
localVarHeaderParameter['Accept'] = 'application/json';
|
||||
|
||||
setSearchParams(localVarUrlObj, localVarQueryParameter);
|
||||
let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {};
|
||||
localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers};
|
||||
|
||||
return {
|
||||
url: toPathString(localVarUrlObj),
|
||||
options: localVarRequestOptions,
|
||||
};
|
||||
},
|
||||
/**
|
||||
* Every record of one log of the run, in `seq` order, unchanged.
|
||||
* @summary List Petri Records
|
||||
|
|
@ -1246,6 +1342,20 @@ export const RunInternalsApiAxiosParamCreator = function (configuration?: Config
|
|||
export const RunInternalsApiFp = function(configuration?: Configuration) {
|
||||
const localVarAxiosParamCreator = RunInternalsApiAxiosParamCreator(configuration)
|
||||
return {
|
||||
/**
|
||||
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
|
||||
* @summary Append Petri Platform Record
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
async appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise<PetriPlatformRecord>> {
|
||||
const localVarAxiosArgs = await localVarAxiosParamCreator.appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options);
|
||||
const localVarOperationServerIndex = configuration?.serverIndex ?? 0;
|
||||
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.appendPetriPlatformRecord']?.[localVarOperationServerIndex]?.url;
|
||||
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
|
||||
},
|
||||
/**
|
||||
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
|
||||
* @summary Append Petri Records
|
||||
|
|
@ -1389,6 +1499,20 @@ export const RunInternalsApiFp = function(configuration?: Configuration) {
|
|||
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.getStageArtifact']?.[localVarOperationServerIndex]?.url;
|
||||
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
|
||||
},
|
||||
/**
|
||||
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
|
||||
* @summary List Petri Platform Records
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {string} [kind] Only the platform records of this kind, as its `kind` tag spells it (`checkpoint`, `pull_request.created`, ...).
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
async listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise<PetriPlatformRecordList>> {
|
||||
const localVarAxiosArgs = await localVarAxiosParamCreator.listPetriPlatformRecords(id, kind, options);
|
||||
const localVarOperationServerIndex = configuration?.serverIndex ?? 0;
|
||||
const localVarOperationServerBasePath = operationServerMap['RunInternalsApi.listPetriPlatformRecords']?.[localVarOperationServerIndex]?.url;
|
||||
return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath);
|
||||
},
|
||||
/**
|
||||
* Every record of one log of the run, in `seq` order, unchanged.
|
||||
* @summary List Petri Records
|
||||
|
|
@ -1615,6 +1739,17 @@ export const RunInternalsApiFp = function(configuration?: Configuration) {
|
|||
export const RunInternalsApiFactory = function (configuration?: Configuration, basePath?: string, axios?: AxiosInstance) {
|
||||
const localVarFp = RunInternalsApiFp(configuration)
|
||||
return {
|
||||
/**
|
||||
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
|
||||
* @summary Append Petri Platform Record
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig): AxiosPromise<PetriPlatformRecord> {
|
||||
return localVarFp.appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options).then((request) => request(axios, basePath));
|
||||
},
|
||||
/**
|
||||
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
|
||||
* @summary Append Petri Records
|
||||
|
|
@ -1728,6 +1863,17 @@ export const RunInternalsApiFactory = function (configuration?: Configuration, b
|
|||
getStageArtifact(id: string, stageId: string, filename: string, retry: number, options?: RawAxiosRequestConfig): AxiosPromise<File> {
|
||||
return localVarFp.getStageArtifact(id, stageId, filename, retry, options).then((request) => request(axios, basePath));
|
||||
},
|
||||
/**
|
||||
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
|
||||
* @summary List Petri Platform Records
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {string} [kind] Only the platform records of this kind, as its `kind` tag spells it (`checkpoint`, `pull_request.created`, ...).
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig): AxiosPromise<PetriPlatformRecordList> {
|
||||
return localVarFp.listPetriPlatformRecords(id, kind, options).then((request) => request(axios, basePath));
|
||||
},
|
||||
/**
|
||||
* Every record of one log of the run, in `seq` order, unchanged.
|
||||
* @summary List Petri Records
|
||||
|
|
@ -1907,6 +2053,18 @@ export const RunInternalsApiFactory = function (configuration?: Configuration, b
|
|||
* RunInternalsApi - object-oriented interface
|
||||
*/
|
||||
export class RunInternalsApi extends BaseAPI {
|
||||
/**
|
||||
* Stores one platform record at the run\'s next `seq`, tied to the Petri stage named by `execution` and `firing` when it belongs to one. The record is the JSON of a Fabro platform record, tagged by `kind`.
|
||||
* @summary Append Petri Platform Record
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {PetriPlatformRecordAppendRequest} petriPlatformRecordAppendRequest
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
public appendPetriPlatformRecord(id: string, petriPlatformRecordAppendRequest: PetriPlatformRecordAppendRequest, options?: RawAxiosRequestConfig) {
|
||||
return RunInternalsApiFp(this.configuration).appendPetriPlatformRecord(id, petriPlatformRecordAppendRequest, options).then((request) => request(this.axios, this.basePath));
|
||||
}
|
||||
|
||||
/**
|
||||
* Appends one batch of records to one log at the sequences they carry, durably, in one transaction. A record equal to the one already stored at its `seq` is accepted without a second append, so a batch whose reply was lost is safe to resend. A different record at a taken `seq`, or a `seq` past the log\'s end, is refused with `petri_record_conflict` and the batch stores nothing.
|
||||
* @summary Append Petri Records
|
||||
|
|
@ -2030,6 +2188,18 @@ export class RunInternalsApi extends BaseAPI {
|
|||
return RunInternalsApiFp(this.configuration).getStageArtifact(id, stageId, filename, retry, options).then((request) => request(this.axios, this.basePath));
|
||||
}
|
||||
|
||||
/**
|
||||
* The run\'s platform records (Fabro\'s own facts about a Petri run: a checkpoint commit, a pull request, a notification), in `seq` order, optionally of one kind. What a run\'s worker reads to find an effect it already performed before performing it again.
|
||||
* @summary List Petri Platform Records
|
||||
* @param {string} id Unique run identifier (ULID).
|
||||
* @param {string} [kind] Only the platform records of this kind, as its `kind` tag spells it (`checkpoint`, `pull_request.created`, ...).
|
||||
* @param {*} [options] Override http request option.
|
||||
* @throws {RequiredError}
|
||||
*/
|
||||
public listPetriPlatformRecords(id: string, kind?: string, options?: RawAxiosRequestConfig) {
|
||||
return RunInternalsApiFp(this.configuration).listPetriPlatformRecords(id, kind, options).then((request) => request(this.axios, this.basePath));
|
||||
}
|
||||
|
||||
/**
|
||||
* Every record of one log of the run, in `seq` order, unchanged.
|
||||
* @summary List Petri Records
|
||||
|
|
|
|||
|
|
@ -269,6 +269,9 @@ export * from './petri-access';
|
|||
export * from './petri-append-request';
|
||||
export * from './petri-open-request';
|
||||
export * from './petri-open-response';
|
||||
export * from './petri-platform-record';
|
||||
export * from './petri-platform-record-append-request';
|
||||
export * from './petri-platform-record-list';
|
||||
export * from './petri-record';
|
||||
export * from './petri-record-list';
|
||||
export * from './petri-release-request';
|
||||
|
|
|
|||
33
lib/packages/fabro-api-client/src/models/petri-platform-record-append-request.ts
generated
Normal file
33
lib/packages/fabro-api-client/src/models/petri-platform-record-append-request.ts
generated
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
/* tslint:disable */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Fabro Run API
|
||||
* HTTP API for managing Fabro workflow run executions.
|
||||
*
|
||||
* The version of the OpenAPI document: 0.2.0
|
||||
*
|
||||
*
|
||||
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
|
||||
* https://openapi-generator.tech
|
||||
* Do not edit the class manually.
|
||||
*/
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* One platform record to store for the run.
|
||||
*/
|
||||
export interface PetriPlatformRecordAppendRequest {
|
||||
/**
|
||||
* The platform record, tagged by `kind`.
|
||||
*/
|
||||
'record': { [key: string]: any; };
|
||||
/**
|
||||
* The Petri execution the record belongs to, with `firing`.
|
||||
*/
|
||||
'execution'?: number;
|
||||
/**
|
||||
* The Petri firing the record belongs to, with `execution`.
|
||||
*/
|
||||
'firing'?: number;
|
||||
}
|
||||
25
lib/packages/fabro-api-client/src/models/petri-platform-record-list.ts
generated
Normal file
25
lib/packages/fabro-api-client/src/models/petri-platform-record-list.ts
generated
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
/* tslint:disable */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Fabro Run API
|
||||
* HTTP API for managing Fabro workflow run executions.
|
||||
*
|
||||
* The version of the OpenAPI document: 0.2.0
|
||||
*
|
||||
*
|
||||
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
|
||||
* https://openapi-generator.tech
|
||||
* Do not edit the class manually.
|
||||
*/
|
||||
|
||||
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
import type { PetriPlatformRecord } from './petri-platform-record';
|
||||
|
||||
/**
|
||||
* The run\'s platform records, in `seq` order.
|
||||
*/
|
||||
export interface PetriPlatformRecordList {
|
||||
'records': Array<PetriPlatformRecord>;
|
||||
}
|
||||
41
lib/packages/fabro-api-client/src/models/petri-platform-record.ts
generated
Normal file
41
lib/packages/fabro-api-client/src/models/petri-platform-record.ts
generated
Normal file
|
|
@ -0,0 +1,41 @@
|
|||
/* tslint:disable */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Fabro Run API
|
||||
* HTTP API for managing Fabro workflow run executions.
|
||||
*
|
||||
* The version of the OpenAPI document: 0.2.0
|
||||
*
|
||||
*
|
||||
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
|
||||
* https://openapi-generator.tech
|
||||
* Do not edit the class manually.
|
||||
*/
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* One of Fabro\'s platform records of a Petri run, as stored: the record\'s JSON tagged by `kind`, its position in the run\'s platform record sequence, and the Petri stage it belongs to when it belongs to one.
|
||||
*/
|
||||
export interface PetriPlatformRecord {
|
||||
/**
|
||||
* The record\'s position in the run\'s platform records, from 1.
|
||||
*/
|
||||
'seq': number;
|
||||
/**
|
||||
* Milliseconds since the Unix epoch when the record was stored.
|
||||
*/
|
||||
'recorded_at': number;
|
||||
/**
|
||||
* The platform record itself, tagged by `kind`.
|
||||
*/
|
||||
'record': { [key: string]: any; };
|
||||
/**
|
||||
* The Petri execution the record belongs to, with `firing`.
|
||||
*/
|
||||
'execution'?: number;
|
||||
/**
|
||||
* The Petri firing the record belongs to, with `execution`.
|
||||
*/
|
||||
'firing'?: number;
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue