Extract workflow E2E tests into workflow/ directory

Move the 6 parametrized workflow scenarios from scenario/workflows.rs
into a new workflow/ directory with one file per test. Move fixture
.fabro files from test/scenario/ to workflow/fixtures/ co-located with
the tests.

Rename the scenario_tests! macro to sandbox_tests! in the new module
for clarity. Slim scenario/ down to just lifecycle and exec tests.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Bryan Helmkamp 2026-03-30 11:19:03 -04:00
parent ef46c174b8
commit 8da8298ea5
No known key found for this signature in database
16 changed files with 390 additions and 365 deletions

View file

@ -1,2 +1,3 @@
mod cmd;
mod scenario;
mod workflow;

View file

@ -1,19 +1,14 @@
mod exec;
mod lifecycle;
mod workflows;
use std::path::{Path, PathBuf};
use std::time::Duration;
use serde_json::Value;
// ---------------------------------------------------------------------------
// Shared helpers
// ---------------------------------------------------------------------------
pub(super) fn fixture(name: &str) -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../../../test/scenario")
.join("tests/it/workflow/fixtures")
.join(name)
}
@ -24,77 +19,6 @@ pub(super) fn read_json(path: &Path) -> Value {
.unwrap_or_else(|e| panic!("failed to parse {}: {e}", path.display()))
}
pub(super) fn read_checkpoint(run_dir: &Path) -> Value {
read_json(&run_dir.join("checkpoint.json"))
}
pub(super) fn read_conclusion(run_dir: &Path) -> Value {
read_json(&run_dir.join("conclusion.json"))
}
/// Find the single run directory under `storage_dir/runs/`.
pub(super) fn find_run_dir(storage_dir: &Path) -> PathBuf {
let runs_base = storage_dir.join("runs");
let entries: Vec<_> = std::fs::read_dir(&runs_base)
.unwrap_or_else(|e| panic!("failed to read {}: {e}", runs_base.display()))
.filter_map(|e| e.ok())
.filter(|e| e.path().is_dir())
.collect();
assert_eq!(
entries.len(),
1,
"expected exactly one run directory under {}",
runs_base.display()
);
entries[0].path()
}
pub(super) fn completed_nodes(run_dir: &Path) -> Vec<String> {
let cp = read_checkpoint(run_dir);
cp["completed_nodes"]
.as_array()
.expect("completed_nodes should be an array")
.iter()
.map(|v| v.as_str().unwrap().to_string())
.collect()
}
pub(super) fn has_event(run_dir: &Path, event_name: &str) -> bool {
let path = run_dir.join("progress.jsonl");
let content = std::fs::read_to_string(&path)
.unwrap_or_else(|e| panic!("failed to read progress.jsonl: {e}"));
content.lines().any(|line| {
if let Ok(v) = serde_json::from_str::<Value>(line) {
v["event"].as_str() == Some(event_name)
} else {
false
}
})
}
// ---------------------------------------------------------------------------
// Macro: generate local_* and daytona_* variants for each scenario
// ---------------------------------------------------------------------------
macro_rules! scenario_tests {
($name:ident) => {
paste::paste! {
#[test]
#[ignore = "scenario: requires local sandbox"]
fn [<local_ $name>]() {
[<scenario_ $name>]("local");
}
#[test]
#[ignore = "scenario: requires DAYTONA_API_KEY"]
fn [<daytona_ $name>]() {
[<scenario_ $name>]("daytona");
}
}
};
}
pub(super) use scenario_tests;
pub(super) fn timeout_for(sandbox: &str) -> Duration {
match sandbox {
"daytona" => Duration::from_secs(600),

View file

@ -1,288 +0,0 @@
use fabro_test::test_context;
use super::{
completed_nodes, find_run_dir, fixture, has_event, read_conclusion, read_json, scenario_tests,
timeout_for,
};
// 1. command_pipeline — two command nodes in sequence, no LLM
scenario_tests!(command_pipeline);
fn scenario_command_pipeline(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.validate()
.arg(fixture("command_pipeline.fabro"))
.assert()
.success();
context
.run_cmd()
.args(["--auto-approve", "--no-retro", "--sandbox", sandbox])
.arg(fixture("command_pipeline.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(
conclusion["status"].as_str(),
Some("success"),
"conclusion status should be success"
);
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"step1".to_string()),
"step1 should be completed"
);
assert!(
nodes.contains(&"step2".to_string()),
"step2 should be completed"
);
// Verify step1 stdout
let stdout1 = std::fs::read_to_string(run_dir.join("nodes/step1/stdout.log"))
.expect("step1 stdout.log should exist");
assert!(
stdout1.contains("hello-from-step1"),
"step1 stdout should contain hello-from-step1, got: {stdout1}"
);
}
// 2. conditional_branching — command + diamond gate, success path taken
scenario_tests!(conditional_branching);
fn scenario_conditional_branching(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args(["--auto-approve", "--no-retro", "--sandbox", sandbox])
.arg(fixture("conditional_branching.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"passed".to_string()),
"passed node should be in completed_nodes: {nodes:?}"
);
assert!(
!nodes.contains(&"failed".to_string()),
"failed node should NOT be in completed_nodes: {nodes:?}"
);
}
// 3. agent_linear — single agent node with LLM
scenario_tests!(agent_linear);
fn scenario_agent_linear(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("agent_linear.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"work".to_string()),
"work should be completed"
);
// Agent node should produce prompt.md and response.md
let prompt_path = run_dir.join("nodes/work/prompt.md");
assert!(prompt_path.exists(), "nodes/work/prompt.md should exist");
let response_path = run_dir.join("nodes/work/response.md");
assert!(
response_path.exists(),
"nodes/work/response.md should exist"
);
let response = std::fs::read_to_string(&response_path).unwrap();
assert!(!response.is_empty(), "response.md should not be empty");
}
// 4. human_gate — human gate with --auto-approve selects first edge
scenario_tests!(human_gate);
fn scenario_human_gate(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("human_gate.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"ship".to_string()),
"ship should be in completed_nodes (auto-approve picks first edge): {nodes:?}"
);
assert!(
!nodes.contains(&"revise".to_string()),
"revise should NOT be in completed_nodes: {nodes:?}"
);
}
// 5. command_agent_mixed — command writes file, agent reads it, command verifies
scenario_tests!(command_agent_mixed);
fn scenario_command_agent_mixed(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("command_agent_mixed.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"setup".to_string()),
"setup should be completed"
);
assert!(
nodes.contains(&"work".to_string()),
"work should be completed"
);
assert!(
nodes.contains(&"verify".to_string()),
"verify should be completed"
);
// Verify command node saw the flag
let stdout = std::fs::read_to_string(run_dir.join("nodes/verify/stdout.log"))
.expect("verify stdout.log should exist");
assert!(
stdout.contains("SCENARIO_FLAG_42"),
"verify stdout should contain SCENARIO_FLAG_42, got: {stdout}"
);
}
// 6. full_stack — command + agent + human gate + goal_gate, kitchen sink
scenario_tests!(full_stack);
fn scenario_full_stack(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("full_stack.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(
conclusion["status"].as_str(),
Some("success"),
"conclusion: {conclusion}"
);
assert!(
conclusion["duration_ms"].as_u64().unwrap_or(0) > 0,
"duration_ms should be > 0"
);
// RunRecord should have key fields
let run_record = read_json(&run_dir.join("run.json"));
assert!(
run_record["run_id"].as_str().is_some(),
"run record should have run_id"
);
assert!(
run_record["graph"]["name"].as_str().is_some(),
"run record should have graph.name"
);
// Progress events
assert!(
has_event(&run_dir, "WorkflowRunStarted"),
"progress should contain WorkflowRunStarted"
);
assert!(
has_event(&run_dir, "WorkflowRunCompleted"),
"progress should contain WorkflowRunCompleted"
);
// All expected nodes completed
let nodes = completed_nodes(&run_dir);
for expected in &["setup", "plan", "approve", "impl", "verify"] {
assert!(
nodes.contains(&expected.to_string()),
"{expected} should be in completed_nodes: {nodes:?}"
);
}
// Verify node stdout should contain PASS
let stdout = std::fs::read_to_string(run_dir.join("nodes/verify/stdout.log"))
.expect("verify stdout.log should exist");
assert!(
stdout.contains("PASS"),
"verify stdout should contain PASS, got: {stdout}"
);
}

View file

@ -0,0 +1,46 @@
use fabro_test::test_context;
use super::{completed_nodes, find_run_dir, fixture, read_conclusion, sandbox_tests, timeout_for};
sandbox_tests!(agent_linear);
fn scenario_agent_linear(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("agent_linear.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"work".to_string()),
"work should be completed"
);
let prompt_path = run_dir.join("nodes/work/prompt.md");
assert!(prompt_path.exists(), "nodes/work/prompt.md should exist");
let response_path = run_dir.join("nodes/work/response.md");
assert!(
response_path.exists(),
"nodes/work/response.md should exist"
);
let response = std::fs::read_to_string(&response_path).unwrap();
assert!(!response.is_empty(), "response.md should not be empty");
}

View file

@ -0,0 +1,50 @@
use fabro_test::test_context;
use super::{completed_nodes, find_run_dir, fixture, read_conclusion, sandbox_tests, timeout_for};
sandbox_tests!(command_agent_mixed);
fn scenario_command_agent_mixed(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("command_agent_mixed.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"setup".to_string()),
"setup should be completed"
);
assert!(
nodes.contains(&"work".to_string()),
"work should be completed"
);
assert!(
nodes.contains(&"verify".to_string()),
"verify should be completed"
);
let stdout = std::fs::read_to_string(run_dir.join("nodes/verify/stdout.log"))
.expect("verify stdout.log should exist");
assert!(
stdout.contains("SCENARIO_FLAG_42"),
"verify stdout should contain SCENARIO_FLAG_42, got: {stdout}"
);
}

View file

@ -0,0 +1,49 @@
use fabro_test::test_context;
use super::{completed_nodes, find_run_dir, fixture, read_conclusion, sandbox_tests, timeout_for};
sandbox_tests!(command_pipeline);
fn scenario_command_pipeline(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.validate()
.arg(fixture("command_pipeline.fabro"))
.assert()
.success();
context
.run_cmd()
.args(["--auto-approve", "--no-retro", "--sandbox", sandbox])
.arg(fixture("command_pipeline.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(
conclusion["status"].as_str(),
Some("success"),
"conclusion status should be success"
);
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"step1".to_string()),
"step1 should be completed"
);
assert!(
nodes.contains(&"step2".to_string()),
"step2 should be completed"
);
let stdout1 = std::fs::read_to_string(run_dir.join("nodes/step1/stdout.log"))
.expect("step1 stdout.log should exist");
assert!(
stdout1.contains("hello-from-step1"),
"step1 stdout should contain hello-from-step1, got: {stdout1}"
);
}

View file

@ -0,0 +1,32 @@
use fabro_test::test_context;
use super::{completed_nodes, find_run_dir, fixture, read_conclusion, sandbox_tests, timeout_for};
sandbox_tests!(conditional_branching);
fn scenario_conditional_branching(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args(["--auto-approve", "--no-retro", "--sandbox", sandbox])
.arg(fixture("conditional_branching.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"passed".to_string()),
"passed node should be in completed_nodes: {nodes:?}"
);
assert!(
!nodes.contains(&"failed".to_string()),
"failed node should NOT be in completed_nodes: {nodes:?}"
);
}

View file

@ -0,0 +1,78 @@
use fabro_test::test_context;
use super::{
completed_nodes, find_run_dir, fixture, has_event, read_conclusion, read_json, sandbox_tests,
timeout_for,
};
sandbox_tests!(full_stack);
fn scenario_full_stack(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("full_stack.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(
conclusion["status"].as_str(),
Some("success"),
"conclusion: {conclusion}"
);
assert!(
conclusion["duration_ms"].as_u64().unwrap_or(0) > 0,
"duration_ms should be > 0"
);
// RunRecord should have key fields
let run_record = read_json(&run_dir.join("run.json"));
assert!(
run_record["run_id"].as_str().is_some(),
"run record should have run_id"
);
assert!(
run_record["graph"]["name"].as_str().is_some(),
"run record should have graph.name"
);
// Progress events
assert!(
has_event(&run_dir, "WorkflowRunStarted"),
"progress should contain WorkflowRunStarted"
);
assert!(
has_event(&run_dir, "WorkflowRunCompleted"),
"progress should contain WorkflowRunCompleted"
);
// All expected nodes completed
let nodes = completed_nodes(&run_dir);
for expected in &["setup", "plan", "approve", "impl", "verify"] {
assert!(
nodes.contains(&expected.to_string()),
"{expected} should be in completed_nodes: {nodes:?}"
);
}
// Verify node stdout should contain PASS
let stdout = std::fs::read_to_string(run_dir.join("nodes/verify/stdout.log"))
.expect("verify stdout.log should exist");
assert!(
stdout.contains("PASS"),
"verify stdout should contain PASS, got: {stdout}"
);
}

View file

@ -0,0 +1,39 @@
use fabro_test::test_context;
use super::{completed_nodes, find_run_dir, fixture, read_conclusion, sandbox_tests, timeout_for};
sandbox_tests!(human_gate);
fn scenario_human_gate(sandbox: &str) {
dotenvy::dotenv().ok();
let context = test_context!();
context
.run_cmd()
.args([
"--auto-approve",
"--no-retro",
"--sandbox",
sandbox,
"--model",
"claude-haiku-4-5",
])
.arg(fixture("human_gate.fabro"))
.timeout(timeout_for(sandbox))
.assert()
.success();
let run_dir = find_run_dir(&context.storage_dir);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("success"));
let nodes = completed_nodes(&run_dir);
assert!(
nodes.contains(&"ship".to_string()),
"ship should be in completed_nodes (auto-approve picks first edge): {nodes:?}"
);
assert!(
!nodes.contains(&"revise".to_string()),
"revise should NOT be in completed_nodes: {nodes:?}"
);
}

View file

@ -0,0 +1,94 @@
mod agent_linear;
mod command_agent_mixed;
mod command_pipeline;
mod conditional_branching;
mod full_stack;
mod human_gate;
use std::path::{Path, PathBuf};
use std::time::Duration;
use serde_json::Value;
pub(super) fn fixture(name: &str) -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.join("tests/it/workflow/fixtures")
.join(name)
}
pub(super) fn read_json(path: &Path) -> Value {
let content = std::fs::read_to_string(path)
.unwrap_or_else(|e| panic!("failed to read {}: {e}", path.display()));
serde_json::from_str(&content)
.unwrap_or_else(|e| panic!("failed to parse {}: {e}", path.display()))
}
pub(super) fn read_conclusion(run_dir: &Path) -> Value {
read_json(&run_dir.join("conclusion.json"))
}
pub(super) fn completed_nodes(run_dir: &Path) -> Vec<String> {
let cp = read_json(&run_dir.join("checkpoint.json"));
cp["completed_nodes"]
.as_array()
.expect("completed_nodes should be an array")
.iter()
.map(|v| v.as_str().unwrap().to_string())
.collect()
}
pub(super) fn has_event(run_dir: &Path, event_name: &str) -> bool {
let path = run_dir.join("progress.jsonl");
let content = std::fs::read_to_string(&path)
.unwrap_or_else(|e| panic!("failed to read progress.jsonl: {e}"));
content.lines().any(|line| {
if let Ok(v) = serde_json::from_str::<Value>(line) {
v["event"].as_str() == Some(event_name)
} else {
false
}
})
}
/// Find the single run directory under `storage_dir/runs/`.
pub(super) fn find_run_dir(storage_dir: &Path) -> PathBuf {
let runs_base = storage_dir.join("runs");
let entries: Vec<_> = std::fs::read_dir(&runs_base)
.unwrap_or_else(|e| panic!("failed to read {}: {e}", runs_base.display()))
.filter_map(|e| e.ok())
.filter(|e| e.path().is_dir())
.collect();
assert_eq!(
entries.len(),
1,
"expected exactly one run directory under {}",
runs_base.display()
);
entries[0].path()
}
macro_rules! sandbox_tests {
($name:ident) => {
paste::paste! {
#[test]
#[ignore = "scenario: requires local sandbox"]
fn [<local_ $name>]() {
[<scenario_ $name>]("local");
}
#[test]
#[ignore = "scenario: requires DAYTONA_API_KEY"]
fn [<daytona_ $name>]() {
[<scenario_ $name>]("daytona");
}
}
};
}
pub(super) use sandbox_tests;
pub(super) fn timeout_for(sandbox: &str) -> Duration {
match sandbox {
"daytona" => Duration::from_secs(600),
_ => Duration::from_secs(180),
}
}