mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-07 03:00:29 +00:00
## Summary
Replaces the `[run.sandbox]` configuration surface with a named,
provider-explicit environment catalog. Runs now select an environment by
slug (`[run.environment] id = "..."`) rather than configuring a sandbox
inline. Fabro resolves the catalog through normal settings precedence,
applies sparse run-level overrides, and creates a concrete sandbox from
the resolved environment.
This is a clean break — no `[run.sandbox]` compatibility layer.
### Plan Summary
- **New config shape:** Top-level `[environments.<slug>]` catalog valid
in `settings.toml`, `.fabro/project.toml`, and `workflow.toml`. Runs
reference a slug via `[run.environment] id = "..."` with optional sparse
overrides under `[run.environment.*]`.
- **Unified environment fields:** `provider`, `image` (ref +
dockerfile), `resources` (cpu/memory/disk), `network` (mode + allow
CIDRs), `lifecycle` (preserve/stop_on_terminal/auto_stop), `labels`,
`volumes`, `env` — replacing the previous split between `[run.sandbox]`,
`[run.sandbox.docker]`, `[run.sandbox.daytona]`, and
`[run.sandbox.daytona.snapshot]`.
- **OpenAPI schema update:** `RunSandboxSettings`, `DockerSettings`,
`DaytonaSettings`, and `DaytonaNetworkLayer` replaced with
`RunEnvironmentSettings`, `EnvironmentSettings`, `EnvironmentProvider`,
`EnvironmentImageSettings`, `EnvironmentResourcesSettings`,
`EnvironmentNetworkSettings`, `EnvironmentLifecycleSettings`, and
`EnvironmentVolumeSettings`.
- **CLI flag rename:** `--sandbox <provider>` → `--environment <slug>`
on `run`, `create`, `preflight`, and `server start/restart`.
- **Provider capability model:** Hard errors for security properties a
provider cannot enforce (local with blocked/CIDR networking; docker with
CIDR allow-lists). Warnings for unsupported resource limits, volumes,
labels, auto-stop, and Docker Dockerfiles.
- **Docs and internal code updated** throughout: `.fabro/project.toml`,
workflow configs, all public docs, CLI args, manifest builders, and the
runner's GitHub credentials check.
### Provider mapping
| Environment field | Local | Docker | Daytona |
|---|---|---|---|
| `image.ref` | Ignored | Docker image | Snapshot name |
| `image.dockerfile` | Ignored | Warning; ignored | Snapshot Dockerfile
(requires `image.ref`) |
| `resources.cpu/memory/disk` | Warning; ignored | cpu_quota / memory
limit / warning | Snapshot sizing |
| `network.mode = block` | **Error** | `network_mode = none` | Daytona
block |
| `network.mode = cidr_allow_list` | **Error** | **Error** | Daytona
CIDR allow-list |
| `labels` | Warning; ignored | Warning; ignored | Daytona labels |
| `volumes` | Warning; ignored | Warning; ignored | Daytona volume
mounts |
| `lifecycle.auto_stop` | Warning; ignored | Warning; ignored | Daytona
auto-stop interval |
| `env` | Process env overlay | Container env | Sandbox env |
### Fabro Details
<details>
<summary>Ran 11 stages in 217m 39s for $129.86</summary>
| Stage | Duration | Cost | Retries |
|---|---|---|---|
| start | 0s | – | 0 |
| toolchain | 1s | – | 0 |
| preflight_compile | 4m 7s | – | 0 |
| preflight_lint | 4m 9s | – | 0 |
| fix_lints | 3m 46s | $1.06 | 0 |
| implement | 76m 6s | $57.39 | 0 |
| simplify_opus | 71m 50s | $38.17 | 0 |
| simplify_gpt | 8m 27s | $2.24 | 0 |
| verify | 6m 10s | – | 0 |
| fixup | 42m 1s | $31.00 | 0 |
| fmt | 3s | – | 0 |
| **Total** | **217m 39s** | **$129.86** | **0** |
</details>
<details>
<summary>Ran <code>ImplementPlan.fabro</code> (12 nodes and 15
edges)</summary>
```dot
digraph ImplementPlan {
graph [
goal="Implement and simplify",
model_stylesheet="
* { model: claude-opus-4-7; }
"
]
rankdir=LR
start [shape=Mdiamond, label="Start"]
exit [shape=Msquare, label="Exit"]
toolchain [label="Toolchain", shape=parallelogram, script="command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", max_retries=0]
preflight_compile [label="Preflight Compile", shape=parallelogram, script="cargo check -q --workspace 2>&1", max_retries=0]
preflight_lint [label="Preflight Lint", shape=parallelogram, script="cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", max_retries=0]
fix_lints [label="Fix Lints", prompt="The preflight lint step failed. Read the build output from context and fix all clippy lint warnings.", max_visits=3]
implement [label="Implement", prompt="Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD.", model="gpt-55", reasoning_effort="xhigh"]
simplify_opus [label="Simplify (Opus)", prompt="@prompts/simplify.md"]
simplify_gpt [label="Simplify (GPT-55)", prompt="@prompts/simplify.md", model="gpt-55"]
verify [label="Verify", shape=parallelogram, script="cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1", goal_gate=true, retry_target="fixup"]
fixup [label="Fixup", prompt="The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors.", max_visits=3]
fmt [label="Format", shape=parallelogram, script="cargo +nightly-2026-04-14 fmt --all 2>&1", max_retries=0]
start -> toolchain
toolchain -> preflight_compile [condition="outcome=succeeded"]
toolchain -> exit
preflight_compile -> preflight_lint [condition="outcome=succeeded"]
preflight_compile -> exit
preflight_lint -> implement [condition="outcome=succeeded"]
preflight_lint -> fix_lints
fix_lints -> preflight_lint
implement -> simplify_opus -> simplify_gpt -> verify
verify -> fmt [condition="outcome=succeeded"]
verify -> fixup
fixup -> verify
fmt -> exit
}
```
</details>
⚒️ Generated with [Fabro](https://fabro.sh)
---------
Co-authored-by: Fabro <noreply@fabro.sh>
Co-authored-by: Bryan Helmkamp <bryan@brynary.com>
Co-authored-by: Bryan Helmkamp <bhelmkamp@users.noreply.github.com>
414 lines
13 KiB
Rust
414 lines
13 KiB
Rust
#![allow(
|
|
clippy::absolute_paths,
|
|
clippy::needless_borrow,
|
|
clippy::needless_borrows_for_generic_args,
|
|
reason = "These workflow-hook tests value explicit fixtures over pedantic style lints."
|
|
)]
|
|
#![expect(
|
|
clippy::disallowed_methods,
|
|
reason = "integration tests stage fixtures with sync std::fs; test infrastructure, not Tokio-hot path"
|
|
)]
|
|
|
|
use std::process::Output;
|
|
|
|
use fabro_config::Storage;
|
|
use fabro_test::{
|
|
TestMode, TwinOpenAi, TwinScenario, TwinScenarios, TwinToolCall, test_context, twin_openai,
|
|
};
|
|
use fabro_vault::{SecretType, Vault};
|
|
|
|
use super::read_conclusion;
|
|
|
|
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
|
|
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
|
|
.await
|
|
.expect("blocking command task should complete")
|
|
}
|
|
|
|
async fn run_failure_output(mut cmd: assert_cmd::Command) -> Output {
|
|
tokio::task::spawn_blocking(move || cmd.assert().failure().get_output().clone())
|
|
.await
|
|
.expect("blocking command task should complete")
|
|
}
|
|
|
|
fn hook_model() -> &'static str {
|
|
if TestMode::from_env().is_twin() {
|
|
"gpt-5.4-mini"
|
|
} else {
|
|
"haiku"
|
|
}
|
|
}
|
|
|
|
fn stage_model() -> &'static str {
|
|
if TestMode::from_env().is_twin() {
|
|
"gpt-5.4-mini"
|
|
} else {
|
|
"claude-haiku-4-5"
|
|
}
|
|
}
|
|
|
|
fn stage_provider() -> &'static str {
|
|
if TestMode::from_env().is_twin() {
|
|
"openai"
|
|
} else {
|
|
"anthropic"
|
|
}
|
|
}
|
|
|
|
fn toml_path(path: &std::path::Path) -> String {
|
|
path.display()
|
|
.to_string()
|
|
.replace('\\', "\\\\")
|
|
.replace('"', "\\\"")
|
|
}
|
|
|
|
fn twin_server_storage_dir(context: &fabro_test::TestContext) -> std::path::PathBuf {
|
|
context.temp_dir.join("hook-server-storage")
|
|
}
|
|
|
|
fn settings_with_hook(context: &fabro_test::TestContext, hook: &str) -> String {
|
|
if TestMode::from_env().is_twin() {
|
|
format!(
|
|
r#"[server.storage]
|
|
root = "{}"
|
|
|
|
[server.auth]
|
|
methods = ["dev-token"]
|
|
|
|
{hook}"#,
|
|
toml_path(&twin_server_storage_dir(context)),
|
|
)
|
|
} else {
|
|
hook.to_string()
|
|
}
|
|
}
|
|
|
|
fn write_hook_settings(context: &fabro_test::TestContext, hook: &str) {
|
|
let settings = settings_with_hook(context, hook);
|
|
if settings.trim().is_empty() {
|
|
return;
|
|
}
|
|
context.write_home(".fabro/settings.toml", settings);
|
|
}
|
|
|
|
fn seed_openai_vault(storage_dir: &std::path::Path, api_key: &str) {
|
|
let mut vault =
|
|
Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load");
|
|
vault
|
|
.set("OPENAI_API_KEY", api_key, SecretType::Token, None)
|
|
.expect("OpenAI credential should store in test vault");
|
|
}
|
|
|
|
fn configure_twin_server(
|
|
context: &mut fabro_test::TestContext,
|
|
_twin: &TwinOpenAi,
|
|
namespace: &str,
|
|
) {
|
|
seed_openai_vault(&twin_server_storage_dir(context), namespace);
|
|
context.isolated_server();
|
|
}
|
|
|
|
fn write_workflow(context: &fabro_test::TestContext, name: &str, dot: &str) -> std::path::PathBuf {
|
|
context.write_temp(name, dot);
|
|
context.temp_dir.join(name)
|
|
}
|
|
|
|
fn configure_hook_env(cmd: &mut assert_cmd::Command, hook_model: &str) {
|
|
cmd.env_remove("CHATGPT_ACCOUNT_ID");
|
|
cmd.env_remove("OPENAI_ORG_ID");
|
|
cmd.env_remove("OPENAI_PROJECT_ID");
|
|
if TestMode::from_env().is_twin() {
|
|
cmd.env_remove("ANTHROPIC_API_KEY");
|
|
}
|
|
cmd.arg("--environment").arg("local");
|
|
cmd.arg("--auto-approve");
|
|
cmd.arg("--provider").arg(stage_provider());
|
|
cmd.arg("--model").arg(hook_model);
|
|
}
|
|
|
|
async fn conclusion_status(context: &fabro_test::TestContext) -> String {
|
|
let run_dir = context.single_run_dir();
|
|
tokio::task::spawn_blocking(move || {
|
|
read_conclusion(&run_dir)["status"]
|
|
.as_str()
|
|
.expect("conclusion should include a string status")
|
|
.to_string()
|
|
})
|
|
.await
|
|
.expect("conclusion status task should complete")
|
|
}
|
|
|
|
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
|
|
async fn hook_prompt_proceed_allows_run() {
|
|
let mut context = test_context!();
|
|
write_hook_settings(
|
|
&context,
|
|
&format!(
|
|
r#"
|
|
[[run.hooks]]
|
|
name = "prompt-proceed"
|
|
event = "run_start"
|
|
prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}."
|
|
model = "{model}"
|
|
"#,
|
|
model = hook_model()
|
|
),
|
|
);
|
|
let workflow = write_workflow(
|
|
&context,
|
|
"hook_prompt_proceed.fabro",
|
|
r"digraph HookTest {
|
|
start [shape=Mdiamond]
|
|
exit [shape=Msquare]
|
|
start -> exit
|
|
}",
|
|
);
|
|
|
|
if TestMode::from_env().is_twin() {
|
|
let twin = twin_openai().await;
|
|
let namespace = format!("{}::{}", module_path!(), line!());
|
|
TwinScenarios::new(namespace.clone())
|
|
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
|
|
.load(twin)
|
|
.await;
|
|
configure_twin_server(&mut context, twin, &namespace);
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
twin.configure_command(&mut cmd, &namespace);
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
} else {
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
}
|
|
|
|
assert_eq!(conclusion_status(&context).await, "succeeded");
|
|
}
|
|
|
|
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
|
|
async fn hook_prompt_block_prevents_run() {
|
|
let mut context = test_context!();
|
|
write_hook_settings(
|
|
&context,
|
|
&format!(
|
|
r#"
|
|
[[run.hooks]]
|
|
name = "prompt-block"
|
|
event = "run_start"
|
|
prompt = "Check: is 2+2 equal to 5? If the statement is true, respond {{\"ok\": true}}. If false, respond {{\"ok\": false, \"reason\": \"math check failed\"}}."
|
|
model = "{model}"
|
|
"#,
|
|
model = hook_model()
|
|
),
|
|
);
|
|
let workflow = write_workflow(
|
|
&context,
|
|
"hook_prompt_block.fabro",
|
|
r"digraph HookTest {
|
|
start [shape=Mdiamond]
|
|
exit [shape=Msquare]
|
|
start -> exit
|
|
}",
|
|
);
|
|
|
|
let output = if TestMode::from_env().is_twin() {
|
|
let twin = twin_openai().await;
|
|
let namespace = format!("{}::{}", module_path!(), line!());
|
|
TwinScenarios::new(namespace.clone())
|
|
.scenario(
|
|
TwinScenario::responses("gpt-5.4-mini")
|
|
.text(r#"{"ok":false,"reason":"math check failed"}"#),
|
|
)
|
|
.load(twin)
|
|
.await;
|
|
configure_twin_server(&mut context, twin, &namespace);
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
twin.configure_command(&mut cmd, &namespace);
|
|
cmd.arg(&workflow);
|
|
run_failure_output(cmd).await
|
|
} else {
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
cmd.arg(&workflow);
|
|
run_failure_output(cmd).await
|
|
};
|
|
|
|
let stderr = String::from_utf8(output.stderr).unwrap();
|
|
assert!(
|
|
stderr.contains("math check failed"),
|
|
"stderr should include hook block reason, got: {stderr}"
|
|
);
|
|
}
|
|
|
|
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
|
|
async fn hook_agent_proceed_allows_run() {
|
|
let mut context = test_context!();
|
|
write_hook_settings(
|
|
&context,
|
|
&format!(
|
|
r#"
|
|
[[run.hooks]]
|
|
name = "agent-proceed"
|
|
event = "run_start"
|
|
prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}. Do not use any tools."
|
|
model = "{model}"
|
|
max_tool_rounds = 1
|
|
agent = "enabled"
|
|
"#,
|
|
model = hook_model()
|
|
),
|
|
);
|
|
let workflow = write_workflow(
|
|
&context,
|
|
"hook_agent_proceed.fabro",
|
|
r"digraph HookTest {
|
|
start [shape=Mdiamond]
|
|
exit [shape=Msquare]
|
|
start -> exit
|
|
}",
|
|
);
|
|
|
|
if TestMode::from_env().is_twin() {
|
|
let twin = twin_openai().await;
|
|
let namespace = format!("{}::{}", module_path!(), line!());
|
|
TwinScenarios::new(namespace.clone())
|
|
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
|
|
.load(twin)
|
|
.await;
|
|
configure_twin_server(&mut context, twin, &namespace);
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
twin.configure_command(&mut cmd, &namespace);
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
} else {
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
}
|
|
|
|
assert_eq!(conclusion_status(&context).await, "succeeded");
|
|
}
|
|
|
|
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
|
|
async fn hook_agent_with_tool_use() {
|
|
let mut context = test_context!();
|
|
let marker = context.temp_dir.join("hook_check.txt");
|
|
std::fs::write(&marker, "READY").unwrap();
|
|
write_hook_settings(
|
|
&context,
|
|
&format!(
|
|
r#"
|
|
[[run.hooks]]
|
|
name = "agent-tools"
|
|
event = "run_start"
|
|
prompt = "Read the file at {path} using the read_file tool. If it contains 'READY', respond with {{\"ok\": true}}. Otherwise respond with {{\"ok\": false, \"reason\": \"not ready\"}}."
|
|
model = "{model}"
|
|
max_tool_rounds = 5
|
|
agent = "enabled"
|
|
"#,
|
|
path = marker.display(),
|
|
model = hook_model()
|
|
),
|
|
);
|
|
let workflow = write_workflow(
|
|
&context,
|
|
"hook_agent_tools.fabro",
|
|
r"digraph HookTest {
|
|
start [shape=Mdiamond]
|
|
exit [shape=Msquare]
|
|
start -> exit
|
|
}",
|
|
);
|
|
|
|
if TestMode::from_env().is_twin() {
|
|
let twin = twin_openai().await;
|
|
let namespace = format!("{}::{}", module_path!(), line!());
|
|
TwinScenarios::new(namespace.clone())
|
|
.scenario(
|
|
TwinScenario::responses("gpt-5.4-mini")
|
|
.tool_call(TwinToolCall::read_file(marker.display().to_string())),
|
|
)
|
|
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
|
|
.load(twin)
|
|
.await;
|
|
configure_twin_server(&mut context, twin, &namespace);
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
twin.configure_command(&mut cmd, &namespace);
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
} else {
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
}
|
|
|
|
assert_eq!(conclusion_status(&context).await, "succeeded");
|
|
}
|
|
|
|
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
|
|
async fn arc_e2e_with_real_llm() {
|
|
let mut context = test_context!();
|
|
write_hook_settings(&context, "");
|
|
let hello = context.temp_dir.join("hello.txt");
|
|
let workflow = write_workflow(
|
|
&context,
|
|
"arc_e2e_real_llm.fabro",
|
|
&format!(
|
|
r#"digraph E2E {{
|
|
graph [goal="Create a test file"]
|
|
start [shape=Mdiamond]
|
|
exit [shape=Msquare]
|
|
work [
|
|
shape=box,
|
|
label="Work",
|
|
prompt="Create a file called hello.txt in {} containing exactly 'Hello from LLM'. Do not output anything else.",
|
|
goal_gate=true
|
|
]
|
|
start -> work -> exit
|
|
}}"#,
|
|
context.temp_dir.display()
|
|
),
|
|
);
|
|
|
|
if TestMode::from_env().is_twin() {
|
|
let twin = twin_openai().await;
|
|
let namespace = format!("{}::{}", module_path!(), line!());
|
|
TwinScenarios::new(namespace.clone())
|
|
.scenario(
|
|
TwinScenario::responses("gpt-5.4-mini")
|
|
.input_contains("Create a file called hello.txt")
|
|
.tool_call(TwinToolCall::write_file(
|
|
hello.display().to_string(),
|
|
"Hello from LLM",
|
|
))
|
|
.text("Done."),
|
|
)
|
|
.load(twin)
|
|
.await;
|
|
configure_twin_server(&mut context, twin, &namespace);
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
twin.configure_command(&mut cmd, &namespace);
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
} else {
|
|
let mut cmd = context.run_cmd();
|
|
configure_hook_env(&mut cmd, stage_model());
|
|
cmd.arg(&workflow);
|
|
run_success_output(cmd).await;
|
|
}
|
|
|
|
assert_eq!(
|
|
std::fs::read_to_string(&hello).unwrap(),
|
|
"Hello from LLM",
|
|
"workflow should create the expected file"
|
|
);
|
|
assert_eq!(conclusion_status(&context).await, "succeeded");
|
|
}
|