Merge origin/main into settings-driven LLM catalog branch

This commit is contained in:
Bryan Helmkamp 2026-05-12 09:00:11 -04:00
commit 52a01a9fa0
No known key found for this signature in database
142 changed files with 15955 additions and 1643 deletions

View file

@ -1 +1 @@
9a3ab8bbba2e71d72ff7703a68e9892e8ce94d8d
a19f6dd03a2ed2b690161476b562462590d78790

View file

@ -1 +1 @@
2f10ee39afe6bd9a1c1df987da52fb82c91b422e
aea71a13e5a9c0c276aff04ccc5e2b71e86ce6d6

View file

@ -15,6 +15,10 @@ leak-timeout = "500ms"
filter = "package(fabro-server) & test(all_spec_routes_are_routable)"
slow-timeout = { period = "15s", terminate-after = 4 }
[[profile.default.overrides]]
filter = "package(fabro-devcontainer) & test(resolve_features_integration)"
slow-timeout = { period = "10s", terminate-after = 3 }
[[profile.default.overrides]]
filter = "package(fabro-workflow)"
slow-timeout = { period = "2s", terminate-after = 3 }

View file

@ -11,7 +11,7 @@ auto_stop_interval = 30
repo = "fabro-sh/fabro"
[run.sandbox.daytona.snapshot]
name = "fabro-v8"
name = "fabro-v9"
cpu = 8
memory = "16GB"
disk = "20GB"
@ -20,6 +20,8 @@ FROM ubuntu:24.04
RUN apt-get update && apt-get install -y --no-install-recommends \
curl git ca-certificates build-essential pkg-config libssl-dev unzip python3 \
xvfb xfce4 xfce4-terminal x11vnc novnc dbus-x11 \
libx11-6 libxrandr2 libxext6 libxrender1 libxfixes3 libxss1 libxtst6 libxi6 \
&& rm -rf /var/lib/apt/lists/*
# GitHub CLI

View file

@ -0,0 +1,11 @@
digraph DaytonaMedium {
graph [goal="Verify the Daytona daytona-medium sandbox starts with standard tooling", retry_target=exit]
rankdir=LR
start [shape=Mdiamond, label="Start"]
exit [shape=Msquare, label="Exit"]
inspect [label="Inspect Sandbox", shape=parallelogram, goal_gate=true, script="set -e\nprintf 'cwd: '; pwd\nprintf 'user: '; whoami\nprintf 'git: '; git --version\nif command -v python3 >/dev/null; then printf 'python: '; python3 --version; else echo 'python: not installed'; fi\nif command -v node >/dev/null; then printf 'node: '; node --version; else echo 'node: not installed'; fi\nprintf 'top-level files:\\n'; ls -la | sed -n '1,40p'"]
start -> inspect -> exit
}

View file

@ -0,0 +1,10 @@
_version = 1
[workflow]
graph = "workflow.fabro"
[run.sandbox]
provider = "daytona"
[run.sandbox.daytona.snapshot]
name = "daytona-medium"

320
Cargo.lock generated
View file

@ -58,6 +58,72 @@ dependencies = [
"subtle",
]
[[package]]
name = "agent-client-protocol"
version = "0.11.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2af62fb84df2af0f933d8f5fd78b843fa5eb0ec5a48fa1b528c41951d0bbe36c"
dependencies = [
"agent-client-protocol-derive",
"agent-client-protocol-schema",
"anyhow",
"futures",
"futures-concurrency",
"jsonrpcmsg",
"rmcp",
"rustc-hash",
"schemars 1.2.1",
"serde",
"serde_json",
"thiserror 2.0.18",
"tokio",
"tokio-util",
"tracing",
"uuid",
]
[[package]]
name = "agent-client-protocol-derive"
version = "0.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ce42c2d3c048c12897eef2e577dfff1e3355c632c9f1625cc953b9df48b44631"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "agent-client-protocol-schema"
version = "0.12.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "49bae57dad1c28a362fbdcf7bab0583316a02b45a70792109fced55780a3b63c"
dependencies = [
"anyhow",
"derive_more",
"schemars 1.2.1",
"serde",
"serde_json",
"serde_with",
"strum",
"tracing",
]
[[package]]
name = "agent-client-protocol-tokio"
version = "0.11.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0e1572b219f22c4b3be0f20f934c8b6f1d1457126ce72923c4f6608f96153b65"
dependencies = [
"agent-client-protocol",
"futures",
"serde",
"serde_json",
"shell-words",
"tokio",
"tokio-util",
]
[[package]]
name = "ahash"
version = "0.8.12"
@ -594,6 +660,15 @@ version = "0.2.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc0b364ead1874514c8c2855ab558056ebfeb775653e7ae45ff72f28f8f3166c"
[[package]]
name = "bs58"
version = "0.5.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4"
dependencies = [
"tinyvec",
]
[[package]]
name = "bstr"
version = "1.12.1"
@ -1088,16 +1163,6 @@ dependencies = [
"darling_macro 0.14.4",
]
[[package]]
name = "darling"
version = "0.21.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9cdf337090841a411e2a7f3deb9187445851f91b309c0c0a29e05f74a00a48c0"
dependencies = [
"darling_core 0.21.3",
"darling_macro 0.21.3",
]
[[package]]
name = "darling"
version = "0.23.0"
@ -1122,20 +1187,6 @@ dependencies = [
"syn 1.0.109",
]
[[package]]
name = "darling_core"
version = "0.21.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1247195ecd7e3c85f83c8d2a366e4210d588e802133e1e355180a9870b517ea4"
dependencies = [
"fnv",
"ident_case",
"proc-macro2",
"quote",
"strsim 0.11.1",
"syn 2.0.117",
]
[[package]]
name = "darling_core"
version = "0.23.0"
@ -1160,17 +1211,6 @@ dependencies = [
"syn 1.0.109",
]
[[package]]
name = "darling_macro"
version = "0.21.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d38308df82d1080de0afee5d069fa14b0326a88c14f15c5ccda35b4a6c414c81"
dependencies = [
"darling_core 0.21.3",
"quote",
"syn 2.0.117",
]
[[package]]
name = "darling_macro"
version = "0.23.0"
@ -1309,6 +1349,7 @@ dependencies = [
"quote",
"rustc_version",
"syn 2.0.117",
"unicode-xid",
]
[[package]]
@ -1537,9 +1578,31 @@ dependencies = [
"libc",
]
[[package]]
name = "fabro-acp"
version = "0.231.0-nightly.1"
dependencies = [
"agent-client-protocol",
"agent-client-protocol-tokio",
"bytes",
"fabro-model",
"fabro-sandbox",
"fabro-types",
"fabro-util",
"futures",
"serde",
"serde_json",
"tempfile",
"thiserror 2.0.18",
"tokio",
"tokio-util",
"tracing",
"uuid",
]
[[package]]
name = "fabro-agent"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -1578,7 +1641,7 @@ dependencies = [
[[package]]
name = "fabro-api"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"chrono",
"fabro-config",
@ -1599,7 +1662,7 @@ dependencies = [
[[package]]
name = "fabro-auth"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -1623,11 +1686,11 @@ dependencies = [
[[package]]
name = "fabro-build-support"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
[[package]]
name = "fabro-checkpoint"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"chrono",
"fabro-config",
@ -1643,7 +1706,7 @@ dependencies = [
[[package]]
name = "fabro-cli"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"assert_cmd",
@ -1661,6 +1724,7 @@ dependencies = [
"dialoguer",
"dirs",
"dotenvy",
"fabro-acp",
"fabro-agent",
"fabro-api",
"fabro-auth",
@ -1678,7 +1742,9 @@ dependencies = [
"fabro-interview",
"fabro-llm",
"fabro-macros",
"fabro-manifest",
"fabro-mcp",
"fabro-mcp-server",
"fabro-model",
"fabro-oauth",
"fabro-proc",
@ -1741,7 +1807,7 @@ dependencies = [
[[package]]
name = "fabro-client"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"bytes",
@ -1770,7 +1836,7 @@ dependencies = [
[[package]]
name = "fabro-config"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"chrono",
@ -1797,7 +1863,7 @@ dependencies = [
[[package]]
name = "fabro-core"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"async-trait",
"fabro-types",
@ -1812,7 +1878,7 @@ dependencies = [
[[package]]
name = "fabro-dev"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"assert_cmd",
@ -1831,7 +1897,7 @@ dependencies = [
[[package]]
name = "fabro-devcontainer"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"fabro-http",
"fabro-static",
@ -1848,7 +1914,7 @@ dependencies = [
[[package]]
name = "fabro-dump"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"bytes",
@ -1862,7 +1928,7 @@ dependencies = [
[[package]]
name = "fabro-github"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"base64",
@ -1884,7 +1950,7 @@ dependencies = [
[[package]]
name = "fabro-graphviz"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"fabro-types",
@ -1898,7 +1964,7 @@ dependencies = [
[[package]]
name = "fabro-hooks"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"async-trait",
"fabro-agent",
@ -1922,7 +1988,7 @@ dependencies = [
[[package]]
name = "fabro-http"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"fabro-static",
"http",
@ -1932,7 +1998,7 @@ dependencies = [
[[package]]
name = "fabro-install"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"base64",
@ -1947,7 +2013,7 @@ dependencies = [
[[package]]
name = "fabro-interview"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"async-trait",
"dialoguer",
@ -1962,7 +2028,7 @@ dependencies = [
[[package]]
name = "fabro-llm"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -1994,7 +2060,7 @@ dependencies = [
[[package]]
name = "fabro-macros"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"clap",
"fabro-options-metadata",
@ -2003,9 +2069,27 @@ dependencies = [
"syn 2.0.117",
]
[[package]]
name = "fabro-manifest"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"fabro-api",
"fabro-config",
"fabro-github",
"fabro-graphviz",
"fabro-template",
"fabro-types",
"fabro-workflow",
"git2",
"temp-env",
"tempfile",
"toml 0.8.23",
]
[[package]]
name = "fabro-mcp"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"fabro-config",
@ -2019,9 +2103,31 @@ dependencies = [
"tracing",
]
[[package]]
name = "fabro-mcp-server"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"chrono",
"fabro-api",
"fabro-client",
"fabro-config",
"fabro-manifest",
"fabro-server",
"fabro-types",
"fabro-util",
"futures",
"rmcp",
"schemars 1.2.1",
"serde",
"serde_json",
"tokio",
"toml 0.8.23",
]
[[package]]
name = "fabro-model"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"fabro-static",
"insta",
@ -2032,7 +2138,7 @@ dependencies = [
[[package]]
name = "fabro-oauth"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"axum",
@ -2054,7 +2160,7 @@ dependencies = [
[[package]]
name = "fabro-options-metadata"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"serde",
"serde_json",
@ -2062,7 +2168,7 @@ dependencies = [
[[package]]
name = "fabro-proc"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"cc",
"libc",
@ -2071,7 +2177,7 @@ dependencies = [
[[package]]
name = "fabro-redact"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"aho-corasick",
"ref-cast",
@ -2087,7 +2193,7 @@ dependencies = [
[[package]]
name = "fabro-sandbox"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -2130,7 +2236,7 @@ dependencies = [
[[package]]
name = "fabro-server"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -2155,6 +2261,7 @@ dependencies = [
"fabro-interview",
"fabro-llm",
"fabro-macros",
"fabro-manifest",
"fabro-model",
"fabro-proc",
"fabro-redact",
@ -2211,7 +2318,7 @@ dependencies = [
[[package]]
name = "fabro-slack"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"fabro-http",
"fabro-interview",
@ -2232,18 +2339,18 @@ dependencies = [
[[package]]
name = "fabro-spa"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"rust-embed",
]
[[package]]
name = "fabro-static"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
[[package]]
name = "fabro-store"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"async-trait",
"bytes",
@ -2270,7 +2377,7 @@ dependencies = [
[[package]]
name = "fabro-telemetry"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"base64",
@ -2296,7 +2403,7 @@ dependencies = [
[[package]]
name = "fabro-template"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"fabro-util",
@ -2308,7 +2415,7 @@ dependencies = [
[[package]]
name = "fabro-test"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"assert_cmd",
"axum",
@ -2331,7 +2438,7 @@ dependencies = [
[[package]]
name = "fabro-tracker"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-trait",
@ -2345,7 +2452,7 @@ dependencies = [
[[package]]
name = "fabro-types"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"chrono",
"clap",
@ -2366,7 +2473,7 @@ dependencies = [
[[package]]
name = "fabro-util"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"console 0.15.11",
@ -2386,17 +2493,18 @@ dependencies = [
[[package]]
name = "fabro-validate"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"fabro-graphviz",
"fabro-model",
"fabro-types",
"serde",
"thiserror 2.0.18",
]
[[package]]
name = "fabro-vault"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"chrono",
"fabro-types",
@ -2408,7 +2516,7 @@ dependencies = [
[[package]]
name = "fabro-workflow"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"assert_cmd",
@ -2417,6 +2525,7 @@ dependencies = [
"bytes",
"chrono",
"dirs",
"fabro-acp",
"fabro-agent",
"fabro-auth",
"fabro-checkpoint",
@ -2530,6 +2639,12 @@ version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582"
[[package]]
name = "fixedbitset"
version = "0.5.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1d674e81391d1e1ab681a28d99df07927c6d4aa5b027d7da16ba32d1d21ecd99"
[[package]]
name = "flatbuffers"
version = "25.12.19"
@ -2800,6 +2915,19 @@ dependencies = [
"futures-sink",
]
[[package]]
name = "futures-concurrency"
version = "7.7.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "175cd8cca9e1d45b87f18ffa75088f2099e3c4fe5e2f83e42de112560bea8ea6"
dependencies = [
"fixedbitset",
"futures-core",
"futures-lite",
"pin-project",
"smallvec",
]
[[package]]
name = "futures-core"
version = "0.3.32"
@ -2823,6 +2951,19 @@ version = "0.3.32"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718"
[[package]]
name = "futures-lite"
version = "2.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f78e10609fe0e0b3f4157ffab1876319b5b0db102a2c60dc4626306dc46b44ad"
dependencies = [
"fastrand",
"futures-core",
"futures-io",
"parking",
"pin-project-lite",
]
[[package]]
name = "futures-macro"
version = "0.3.32"
@ -3689,6 +3830,16 @@ dependencies = [
"wasm-bindgen",
]
[[package]]
name = "jsonrpcmsg"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6d833a15225c779251e13929203518c2ff26e2fe0f322d584b213f4f4dad37bd"
dependencies = [
"serde",
"serde_json",
]
[[package]]
name = "jsonschema"
version = "0.42.2"
@ -5587,6 +5738,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2231b2c085b371c01bc90c0e6c1cab8834711b6394533375bdbf870b0166d419"
dependencies = [
"async-trait",
"base64",
"chrono",
"futures",
"http",
@ -6134,11 +6286,12 @@ dependencies = [
[[package]]
name = "serde_with"
version = "3.17.0"
version = "3.20.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "381b283ce7bc6b476d903296fb59d0d36633652b633b27f64db4fb46dcbfc3b9"
checksum = "e72c1c2cb7b223fafb600a619537a871c2818583d619401b785e7c0b746ccde2"
dependencies = [
"base64",
"bs58",
"chrono",
"hex",
"indexmap 1.9.3",
@ -6153,11 +6306,11 @@ dependencies = [
[[package]]
name = "serde_with_macros"
version = "3.17.0"
version = "3.20.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a6d4e30573c8cb306ed6ab1dca8423eec9a463ea0e155f45399455e0368b27e0"
checksum = "b90c488738ecb4fb0262f41f43bc40efc5868d9fb744319ddf5f5317f417bfac"
dependencies = [
"darling 0.21.3",
"darling 0.23.0",
"proc-macro2",
"quote",
"syn 2.0.117",
@ -6888,6 +7041,7 @@ checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098"
dependencies = [
"bytes",
"futures-core",
"futures-io",
"futures-sink",
"futures-util",
"hashbrown 0.15.5",
@ -7140,7 +7294,7 @@ dependencies = [
[[package]]
name = "twin-github"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"axum",
"base64",
@ -7159,7 +7313,7 @@ dependencies = [
[[package]]
name = "twin-openai"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
dependencies = [
"anyhow",
"async-stream",

View file

@ -5,10 +5,12 @@ resolver = "2"
[workspace.package]
edition = "2021"
version = "0.230.0-nightly.0"
version = "0.231.0-nightly.1"
license = "MIT"
[workspace.dependencies]
agent-client-protocol = { version = "0.11.1", features = ["unstable_session_usage"] }
agent-client-protocol-tokio = "0.11.1"
anyhow = "1"
axum = { version = "0.8" }
axum-extra = { version = "0.10", features = ["cookie-private"] }

View file

@ -1952,6 +1952,8 @@ Emitted when an image or snapshot ensure step fails.
## CLI ensure events
These legacy events may appear in older run logs. Current CLI backend runs do not emit them because Fabro no longer installs or prepares provider CLIs at stage runtime.
### `cli.ensure.started`
```json

View file

@ -401,7 +401,7 @@ V2 keeps the current durable family surface broadly intact.
- `sandbox.*`
- `setup.*`
- `cli.ensure.*`
- `cli.ensure.*` (legacy only)
- `command.*`
- `agent.cli.*`
- `devcontainer.*`

View file

@ -0,0 +1,246 @@
# Fabro MCP Server — QA Test Plan
One-time manual QA pass for the 5 tools exposed by `fabro-mcp-server`. Source of truth: `lib/crates/fabro-mcp-server/src/run_tools/`.
This plan is **not** a template for adding automated test coverage — it exists to drive a single hands-on sweep against a real running server. Tick boxes as scenarios pass; add notes inline for failures or surprising behavior. Open bugs/PRs for issues found; do not port these scenarios into the Rust test suite.
## Findings rollup
Live list of bugs and notable observations surfaced during the sweep. Each entry links back to the scenario where it was found.
### Bugs / mismatches
None currently open.
### Rechecked / no longer open
- **C4 — `inputs` schema/runtime mismatch**: fixed by narrowing MCP input values to scalar JSON (`string`, `boolean`, `integer`, `number`) and rejecting arrays/objects locally with scalar-only errors. Re-tested on 2026-05-11 against `127.0.0.1:32276`; `tools/list` now advertises scalar-only `inputs.additionalProperties`.
- **C5 — Misleading null-input error message**: fixed. Re-tested on 2026-05-11; null now returns ``input `maybe` cannot be null; use a string, boolean, or number``.
- **I7 / I9 — Misleading "Run not found." on terminal runs**: fixed on 2026-05-11 in the server API layer. `message`/steer against a durable terminal run that no longer has a live managed engine now returns `409` with `run_not_steerable`; `cancel` returns `409` with `Run is already terminal and cannot be cancelled.` True missing runs still return `404`.
- **I10 — Archived runs not filtered from default search**: fixed on 2026-05-11 by aligning MCP search with the HTTP API. `fabro_run_search` now hides archived runs when `archived` is omitted, while `archived=true` still searches archived runs explicitly.
- **I15 / I16 — yes/no answer flow**: re-tested on 2026-05-11 against `fabro server` `0.230.0-nightly.0` at `127.0.0.1:32276`. `answer=true` and `answer=false` both submit successfully for the bundled `interview` workflow's first `yes_no` question. `true` advanced the run to the next `confirmation` question.
- **I22 — numeric answer local validation**: re-tested on 2026-05-11 against the same server. `answer=42` now returns `unsupported answer value: 42; expected boolean, string, or object` from the MCP layer before reaching the API.
- **Section 2 side observation — Search payloads include full `goal` text**: fixed on 2026-05-11. `fabro_run_search` now returns bounded `goal_preview` plus `goal_truncated` instead of the full `goal`, keeping list responses compact while preserving full summaries on other run interactions.
- **X6 — Cursor/filter ordering**: simplified on 2026-05-11 by applying search filters before sorting and applying the `after` cursor. This prevents unrelated runs outside the filtered result set from trimming the page. Pagination is explicitly not snapshot-isolated; a new matching run inserted before the cursor during traversal appears when the client starts a new search.
### UX / polish
- **C12 — `cwd` errors don't distinguish "directory missing" from "workflow not in directory"**: both return `workflow not found: <slug>`.
- **S9 (bonus) — Undocumented date format**: error message reveals `YYYY-MM-DD` is accepted alongside RFC3339, but the schema only says RFC3339.
- **S17 — `run_ids` accepts more than IDs**: error message reveals it also matches ID prefixes and workflow names. Either rename the field or document.
- **E4 — Events `search` is whole-envelope substring match**: search includes embedded payloads (workflow definitions, settings, sandbox dockerfile, etc.), so a search like `query="list_prs"` legitimately matches the `run.created` event because that event embeds the workflow JSON. Easy to misinterpret. Consider documenting or scoping search to event body only.
### Nice-to-haves
- **C16 — Helpful error**: unknown workflow lists available workflows. Keep.
## Pre-flight (all tools)
- [ ] **P1** Server unreachable — stop `fabro server`, call any tool, expect a clear connection-error message (not a panic, not a hang).
- [ ] **P2** Schema discovery — list tools through an MCP client; verify each tool has a complete JSON schema and the documented `anyOf` for `AnswerValue`.
---
## 1. `fabro_run_create`
Source: `run_tools/create.rs:124`
### Happy path
- [x] **C1** Create one run from an existing workflow (e.g. `gh-list`); default `start=true` → expect `started=true`, `status` in `{queued, starting, running}`. — **PASS**. `status=queued`.
- [x] **C2** Create with `start=false` → expect `started=false`, `status=submitted`. — **PASS**. Run `01KRC4MP2NEQS9GJDE9FJ0EECH` kept as fixture for I3.
- [x] **C3** Batch create 5 runs in one call → all return; result preserves array order. — **PASS**. ULIDs monotonically increasing.
### Inputs / manifest
- [x] **C4** Pass `inputs` with string / number / boolean / nested object / array → **PASS** after 2026-05-11 recheck. Scalar values are accepted. Arrays and objects are rejected locally with scalar-only errors, and the MCP schema now advertises scalar-only `inputs` values.
- [x] **C5** `inputs` containing `null` → **PASS** after 2026-05-11 recheck. Returns ``input `maybe` cannot be null; use a string, boolean, or number``.
- [x] **C6** `labels={"team": "qa"}` round-trip via search. — **PASS**. All 5 C3 runs returned with labels intact.
- [x] **C7** Optional flags: `goal`, `model+provider`, `sandbox`, `preserve_sandbox+auto_approve+dry_run`. — **PASS** all accepted; `goal` override round-tripped via search.
- [x] **C8** Custom `run_id`: valid ULID accepted (`01KRC500000000C8TEST00000A`); wrong length → `invalid length`; invalid Crockford char (e.g. `U`) → `invalid character`. — **PASS**.
### `cwd`
- [x] **C9** Omit `cwd` → uses base CWD. — **PASS** (covered by every prior scenario).
- [x] **C10** `cwd` to repo root resolves workflow. — **PASS**.
- [x] **C11** `cwd=/tmp` (no `.fabro/workflows`) → `workflow not found: gh-list`. — **PASS**.
- [x] **C12** `cwd=/this/path/does/not/exist/xyz123` → same generic `workflow not found: gh-list`. — **PASS but note**: error doesn't distinguish "directory missing" from "workflow not in directory". Minor UX gap.
### Validation
- [x] **C13** Empty `runs: []` → `runs must contain at least 1 item(s)`. — **PASS**.
- [x] **C14** 51 entries → `runs must contain no more than 50 item(s)`. — **PASS**.
- [x] **C15** Missing required `workflow` → MCP layer `-32602: missing field 'workflow'`. — **PASS**.
- [x] **C16** Unknown workflow slug → `Unknown workflow 'X'\n\nAvailable workflows: ...`. — **PASS** (very helpful — lists available workflows).
### Failure semantics
- [x] **C17** Invalid sandbox name → `failed to resolve manifest settings: run.sandbox.provider: invalid value - unknown sandbox provider: this-sandbox-does-not-exist`. — **PASS**. Error raised at manifest-resolve time before any run record is created (no orphaned submitted run).
---
## 2. `fabro_run_search`
Source: `run_tools/search.rs:75`
### Happy path
- [x] **S1** No params → returns up to 20 runs, sorted by `started_at OR created_at` desc. — **PASS**. Mixed-timestamp ordering correct (succeeded run at pos 8 sorts by its `started_at` between two `created_at`-only runs).
- [x] **S2** `first=5` → exactly 5; `next_cursor` is the last run's ID. — **PASS**.
- [x] **S3** `first=100` → all 17 runs, `next_cursor=null`. — **PASS**.
- [x] **S4** Cursor follow-through: page 1 IDs `[A, B]`, page 2 with `after=B` returns `[C, D]`. No overlap. — **PASS**. Note: cursors are run IDs, not opaque tokens.
### Filters
- [x] **S5** `workflow="smoke"` (slug) and `workflow="Smoke"` (name) both match same run. — **PASS**.
- [x] **S6** `status=["succeeded"]` → 4; `["failed","dead"]` → 1; `["submitted"]` → 5. — **PASS**.
- [x] **S7** Labels round-trip. — **PASS** (verified via C6).
- [x] **S8** `archived=false` → all unarchived runs; `archived=true` → `[]` (no archived runs yet). Re-verify after I10. — **PARTIAL** (no archived fixtures yet).
- [x] **S9** `created_after`/`created_before` (RFC3339) bound results correctly; tight window `17:00–18:00` returns only old runs. — **PASS**. **Bonus**: error message reveals `YYYY-MM-DD` is also accepted — undocumented in the schema.
- [x] **S10** `run_ids=[A,B,A]` → 2 deduped runs. — **PASS**.
- [x] **S11** Combined `workflow + status + labels + archived` → returns exactly the 5 batch=c3 runs. — **PASS**.
### Validation
- [x] **S12** `first=101` → `first must be <= 100`. — **PASS**.
- [x] **S13** `run_ids=[]` → `run_ids must contain at least 1 item(s)`. — **PASS**.
- [x] **S14** `run_ids` length 101 → `run_ids must contain no more than 100 item(s)`. — **PASS**.
- [x] **S15** `status=["bogus"]` → `unknown run status 'bogus'`. — **PASS**.
- [x] **S16** `created_after="not-a-date"` → `created_after must be RFC3339 or YYYY-MM-DD: input contains invalid characters`. — **PASS**.
### Edge cases
- [x] **S17** Non-existent ID in `run_ids` → `No run found matching '<ID>' (tried run ID prefix and workflow name)`. — **PASS** + **finding**: `run_ids` also accepts ID prefixes and workflow names, which is broader than the field name suggests.
- [x] **S18** No matches → `{"runs": [], "next_cursor": null}`. — **PASS**.
- [x] **S19** Bogus `after=<unknown>` → returns full first page (skip never applies). — **PASS** as documented.
### Side observation
Search responses include the full `goal` text per run; a single `ImplementPlan` run can add ~30 KB to every search payload. Consider truncating `goal` (or excluding it from list responses) the way events have `max_content_length`. **Logged in Findings.**
---
## 3. `fabro_run_gather`
Source: `run_tools/gather.rs:56`
### Happy path
- [x] **G1** Gather 1 already-terminal run → instant return, `timed_out=false`, `elapsed_seconds=0`. — **PASS**.
- [x] **G2** In-flight `gh-list` with `timeout=60, poll=5` → reaches `succeeded`, `timed_out=false`, `elapsed=30`. — **PASS**.
- [x] **G3** In-flight `gh-list` with `timeout=5, poll=5` → `timed_out=true`, `elapsed=5`, run still `starting`. — **PASS**.
- [x] **G4** Mix of 2 terminal + 1 in-flight, `timeout=90, poll=5` → all 3 succeeded, `timed_out=false`, `elapsed=40`. — **PASS**.
### Validation
- [x] **G5** `run_ids=[]` → `run_ids must contain at least 1 item(s)`. — **PASS**.
- [x] **G6** 51 IDs → `run_ids must contain no more than 50 item(s)`. — **PASS**.
- [x] **G7** `timeout_seconds=601` → `timeout_seconds must be <= 600`. — **PASS**.
- [x] **G8** `poll_interval_seconds=4` → `poll_interval_seconds must be >= 5`. — **PASS**.
- [x] **G9** Omit both → call accepted; terminal run still returns instantly. Default values per source: `timeout=300, poll=15`. — **PASS**.
### Edge cases
- [x] **G10** Non-existent run ID → `No run found matching '<ID>' (tried run ID prefix and workflow name)`. — **PASS** (same fuzzy match as search).
- [x] **G11** Poll cadence: G3 confirms last sleep clamps to deadline (`elapsed=5` exactly with `timeout=5, poll=5`). — **PASS** (inferred from G2/G3 timing).
- [ ] **G12** Run cancelled mid-gather → terminal `failed(status_reason=cancelled)` quickly. — **DEFERRED** to after I8 (cancel).
- [ ] **G13** Run becomes `blocked` — verify gather still waits. — **DEFERRED** to after Section 5 (interview workflow).
---
## 4. `fabro_run_events`
Source: `run_tools/events.rs:115`
### Actions
- [x] **E1** `list` no filters → 45 events (`gh-list` has full lifecycle: run.*, sandbox.*, git.*, stage.*, etc.), `next_cursor=46`. — **PASS**.
- [x] **E2** `details` with 2 event_ids → returns exactly those 2 envelopes. — **PASS**.
- [x] **E3** `details` with no `event_ids` → `event_ids is required for details action`. — **PASS**.
- [x] **E4** `search query="list_prs"` → 14 events. Includes `run.created` because it embeds the full workflow definition (which contains the `list_prs` node ID). — **PASS** + **observation**: search ranges over the entire serialized envelope, so big embedded payloads (workflow defs, settings) can produce non-obvious hits.
- [x] **E5** `search` with missing `query` → `query is required for search action`. — **PASS**.
### Filters
- [x] **E6** `event_types=["stage.started"]` → exactly 4 events (start, list_prs, list_issues, exit). — **PASS**.
- [x] **E7** `categories=["git","sandbox"]` → 12 events all with prefix `git.*` or `sandbox.*`. — **PASS**.
- [x] **E8** `created_after=17:03:10Z` + `created_before=17:03:13Z` → 5 events all timestamped 17:03:12.89x. — **PASS**.
- [x] **E9** Combined `event_types + offset + first` covered by E14.
### Pagination & direction
- [x] **E10** Page 1 `first=10` → seqs 1–10, `next_cursor=11`. Page 2 `after=11, first=5` → seqs 11–15, `next_cursor=16`. No duplicates; contiguous. — **PASS**.
- [x] **E11** `direction=desc, first=5` → seqs 45, 44, 43, 42, 41; `next_cursor=41` (last seq, no +1 — per the desc branch). — **PASS**.
- [x] **E12** Default direction = asc (E10 confirms). — **PASS**.
- [x] **E13** `direction="weird"` → `direction must be 'asc' or 'desc'`. — **PASS**.
- [x] **E14** `event_types=["stage.started"], offset=2, first=5` → returned 2 events (seqs 29, 39) — correctly skipped the first 2 (15, 19) of the 4 matching. — **PASS**.
- [x] **E15** `limit=3` → 3 events. — **PASS** (alias works).
### Truncation
- [x] **E16** `stage.completed, first=1, max_content_length=200` → 1 event, `truncated=true`, `event` is a JSON string. — **PASS**.
- [x] **E17** UTF-8 boundary — **VERIFIED via existing unit test** at `events.rs:269-312`. Can't easily reproduce through MCP surface (no multibyte event content in default fixtures).
- [x] **E18** Default `max_content_length=20000` → all 5 events `truncated=false` (including the ~5 KB `run.created`). — **PASS**.
### Validation
- [x] **E19** `run_id=" "` (whitespace) → `run_id is required`. — **PASS**.
- [x] **E20** `first=201` → `first must be <= 200`. — **PASS**.
- [x] **E21** Non-existent run ID → fuzzy-match error (same as search/gather). — **PASS**.
---
## 5. `fabro_run_interact`
Source: `run_tools/interact.rs:201`
### Actions
#### `get`
- [x] **I1** Returns `{summary, projection}`; projection includes `spec`, `graph`, `status`, `checkpoints`, `pending_interviews`, `stages`, `sandbox`, `conclusion`, etc. — **PASS**.
- [x] **I2** Non-existent run → fuzzy match error. — **PASS**.
#### `start`
- [x] **I3** Non-started run (from C2) → `start` transitions to `queued`. Second `start` → `an engine process is still running for this run — cannot start`. — **PASS**.
#### `message` (steer)
- [ ] **I4** Steer a running LLM agent — **DEFERRED** (requires an active LLM agent stage; would burn LLM tokens; can be exercised manually once the answer bug below is resolved).
- [ ] **I5** `interrupt=true` — **DEFERRED** along with I4.
- [x] **I6** Missing `message` → `message is required for action message`. — **PASS**.
- [x] **I7** Message a terminal run → initially returned `Run not found.`. — **FIXED**: durable terminal runs without a live managed engine now return `409 run_not_steerable`; true missing runs remain `404`.
#### `cancel`
- [x] **I8** Cancel a `gh-list` run during `starting`. Returns summary at request time (status=`starting`). Subsequent `gather` returned terminal `failed` within 5s; `get` projection shows `status: {kind: "failed", reason: "cancelled"}` and `conclusion.failure_reason: "Pipeline cancelled"`. — **PASS** + **observation**: `cancel`'s returned summary is a snapshot at request time, not the eventual terminal status.
- [x] **I9** Cancel an already-terminal run → initially returned `Run not found.`. — **FIXED**: durable terminal runs without a live managed engine now return `409` with `Run is already terminal and cannot be cancelled.`; true missing runs remain `404`.
#### `archive` / `unarchive`
- [x] **I10** Archive terminal run → `archived=true` in summary; visible via `search archived=true`. — **FIXED**: default search now hides archived runs to match `/api/v1/runs`; `archived=true` still surfaces archived runs explicitly.
- [x] **I11** Unarchive → reverses (`archived=false`). — **PASS**.
- [x] **I12** Archive an active run → `run <id> must be terminal (succeeded, failed, or dead) to archive; current status is starting`. — **PASS** (excellent error).
#### `get_questions`
- [x] **I13** Terminal run → `questions: []`. — **PASS**.
- [x] **I14** Blocked interview run → returns full question record (id, text, options, question_type, stage, allow_freeform). — **PASS**.
#### `answer` — `AnswerValue` shapes
Re-check note: the earlier `yes_no` answer failure did not reproduce against `fabro server` `0.230.0-nightly.0` on `127.0.0.1:32276` (2026-05-11). Boolean answers are accepted for `yes_no` questions, and invalid question/type combinations are rejected by the API as expected.
- [x] **I15** `answer=true` on the first `yes_no` question → submitted successfully (`submitted=true`) and advanced to the `confirmation` question. — **PASS**. Run `01KRCAQ9AS14KFCW4CXBZQ0CW9`.
- [x] **I16** `answer=false` on a fresh `yes_no` question → submitted successfully (`submitted=true`). — **PASS**. Run `01KRCATZ031CAPEPVB4CNFEE33`.
- [ ] **I17** `answer="some text"` — **NOT RE-TESTED**. Should be tested against a `freeform` question or a question with `allow_freeform=true`; text is not valid for the bundled `yes_no` question.
- [ ] **I18** `answer={"text":"hi"}` — **NOT RE-TESTED**. Same scope as I17.
- [x] **I19** `answer={"option":"Y"}` against the first `yes_no` question → `Answer does not match question type.` — **PASS / expectation corrected**. The MCP layer maps this shape to `selected`, but `server.rs:2670-2710` only accepts `yes`/`no` for `yes_no` and `confirmation`; `selected` belongs to `multiple_choice`.
- [ ] **I20** `answer={"options":[...]}` — **NOT RE-TESTED**. Should be tested against a `multi_select` question; `multi_selected` is not valid for `yes_no`.
- [x] **I21** `answer={"value":"yes"}` → `answer object must contain one of: option, options, text` (local validation). — **PASS**.
- [x] **I22** `answer=42` (number) → `unsupported answer value: 42; expected boolean, string, or object`. — **PASS** (local validation).
- [x] **I23** `answer={"option": 5}` → `answer option must be a string: invalid type: integer '5', expected a string`. — **PASS**.
- [x] **I24** `answer={"options": ["a", 2]}` → `answer options must be strings: invalid type: integer '2', expected a string`. — **PASS**.
- [x] **I25** `action=answer` without `question_id` → `question_id is required for action answer`. — **PASS**.
- [x] **I26** `action=answer` without `answer` → `answer is required for action answer`. — **PASS**.
- [x] **I27** Already-answered question — observed indirectly: the same question_id returned `Question no longer exists or was already answered.` on retry. — **PASS**.
### Cross-cutting
- [x] **I28** `run_id=" "` → `run_id is required`. — **PASS**.
- [x] **I29** Action enum: `Get` and `get-questions` both rejected with `unknown variant 'X', expected one of: get, start, message, cancel, archive, unarchive, get_questions, answer`. — **PASS**.
---
## 6. End-to-end scenarios (multi-tool)
- [x] **X1 — Happy lifecycle** `gh-list` create → 35s gather → events filtered to `stage.started/completed` → 8 events for 4 stages (start, list_prs, list_issues, exit). Sequence matches workflow graph. — **PASS**.
- [x] **X2 — Cancel mid-run** Covered by I8: `gh-list` cancel during `starting` → gather returned terminal `failed` in 5s; projection shows `status_reason=cancelled`. — **PASS**.
- [ ] **X3 — Human-in-the-loop** — **PARTIAL**. The earlier yes/no answer blocker is no longer reproduced (I15/I16 now pass), and `gather` returning `timed_out=true` on a `blocked` run **was** verified (G13). Full interview completion remains unverified in this sweep.
- [ ] **X4 — Steering** — **DEFERRED** (requires active LLM agent).
- [x] **X5 — Archive flow** Covered by I10/I11: archive → search with `archived=true` returns it (also returned by default search — see I10 finding). Unarchive reverses. — **PASS** with caveat.
- [x] **X6 — Search/cursor under churn** Page 1 `first=3` → cursor saved. Created new run `01KRC625KG…` mid-flow. Page 2 with original cursor returned 3 older runs; a fresh page 1 placed the new run at position 1. — **ACCEPTED / SIMPLIFIED**. Pagination is not snapshot-isolated; clients that need newly inserted earlier results should restart the search. Code now applies filters before sorting/cursoring so unrelated runs outside the filtered result set do not trim filtered pages.
- [x] **X7 — Events while running** Started `gh-list` run, listed events `desc` immediately (max seq=8), gathered to completion, re-listed (max seq=46). Seq numbers grew monotonically; no early events lost. — **PASS**.
- [x] **X8 — Truncated event recovery** Fetched the `ImplementPlan` `run.created` event (embeds ~30 KB goal) at default `max_content_length=20000` → `truncated:true`, payload returned as a JSON string. — **PASS**.
- [ ] **X9 — Stranger inputs** — Skipped per scope decision. Trivially safe since inputs go through TOML conversion to be stored as values; the MCP layer never opens paths.
---
## 7. Mechanics for the manual sweep
- **Driver** — run these scenarios through an MCP client (e.g. Claude Code with the `fabro` MCP server configured) against a locally running `fabro server`.
- **Reusable run IDs** — keep a handful of already-terminal runs around (e.g. one `gh-list` succeeded, one failed `implement-plan`) as fixtures for `events`, `gather` (instant-return), `interact.get`, and `archive` scenarios.
- **Server unreachable cases** — stop the API server with the MCP client still connected to exercise error propagation paths.
- **Issue tracking** — file a GitHub issue per defect; link the scenario ID (e.g. `C13`) so this plan and the bugs cross-reference.

View file

@ -0,0 +1,326 @@
# ACP Backend Test Plan
The accepted testing strategy still holds, with scoped additions from the implementation plan: ACP prompt nodes are explicitly supported, backend validation becomes strict, sandbox stdio is part of the public contract, and ACP events affect run projection, fork replay, and server steerability. These additions do not require paid services or materially change the agreed scope because all high-value ACP checks can run against deterministic fake agents and local/unit harnesses.
## Harness Requirements
1. **Fake ACP agent harness**
- What it does: runs a deterministic ACP agent over stdio, records observed JSON-RPC method order, emits configurable `session/update` messages, writes optional files in cwd, responds to permission requests, and simulates cancellation, malformed JSON, early exit, timeout, and stop reasons.
- Exposes: a checked-in fixture binary or script plus crate-local helpers in `fabro-acp::test_support` using `agent-client-protocol` schema types where practical.
- Complexity: medium. It is the main substitute for paid/live ACP agents.
- Tests depending on it: 7, 8, 9, 10, 11, 12, 13, 26, 27.
2. **Sandbox stdio process harness**
- What it does: exercises `Sandbox::spawn_stdio_process` with a line-oriented subprocess, captures stdout/stderr separately, terminates the process, and validates Docker exec option construction without requiring live Docker.
- Exposes: local sandbox round-trip tests, Docker option-builder/control-wrapper tests, Daytona unsupported-provider assertion, and decorator/test-support forwarding assertions.
- Complexity: medium because Docker stdio is multiplexed and cancellation needs an explicit control path.
- Tests depending on it: 5, 6, 8, 12, 26.
3. **Workflow ACP runner harness**
- What it does: runs a real Fabro workflow with `backend="acp"` and node-level `acp_command` pointing at the fake ACP agent, then inspects persisted run events/projection through existing CLI workflow helpers.
- Exposes: user-visible `fabro run` result, run events, stage response, `files_touched`, and `provider_used`.
- Complexity: low once the fake ACP agent exists.
- Tests depending on it: 26, 27.
4. **Server event-state harness extension**
- What it does: uses existing server test fixtures to insert a running run, apply ACP events, and call `POST /runs/{id}/steer`.
- Exposes: HTTP status and JSON error codes through the Axum test router.
- Complexity: low.
- Tests depending on it: 23, 24.
## Test Plan
1. **Existing CLI backend behavior remains intact**
- Type: regression
- Disposition: existing
- Harness: existing `fabro-workflow` router/CLI tests from the accepted strategy.
- Preconditions: current repository before ACP changes; no ACP-specific code required.
- Actions: run `ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_uses_cli_for_backend_attr) | test(router_uses_api_by_default) | test(backend_router_delegates_to_cli_for_cli_node) | test(backend_router_delegates_to_api_for_normal_node) | test(backend_router_delegates_to_cli_for_backend_attr) | test(full_pipeline_with_cli_backend_node) | test(stylesheet_backend_property_routes_to_cli) | test(cli_backend_run_writes_prompt_and_calls_exec) | test(cli_backend_run_with_codex_provider) | test(parse_real_codex_ndjson)'`.
- Expected outcome: all tests pass; `backend="cli"` still routes agent nodes to CLI, default routing remains API, stylesheet `backend: cli` still works, CLI output parsing remains unchanged. Source of truth: user request to keep `api`, `cli`, and `acp` as three backends for now; implementation plan User-Visible Behavior for legacy CLI compatibility.
- Interactions: router, CLI backend, stylesheet import, sandbox command execution, CLI event emission.
2. **Existing stdio JSON-RPC precedent remains intact**
- Type: regression
- Disposition: existing
- Harness: existing `fabro-mcp` stdio integration tests.
- Preconditions: Python is available; no live MCP service required.
- Actions: run `ulimit -n 4096 && cargo nextest run -p fabro-mcp -E 'test(stdio_client_initialize_and_list_tools) | test(stdio_client_call_tool_echo) | test(connection_manager_stdio_roundtrip)'`.
- Expected outcome: all tests pass; Fabro can still spawn a stdio JSON-RPC collaborator, initialize it, list capabilities, and call it. Source of truth: accepted strategy listed these as relevant stdio precedent.
- Interactions: child process stdio, JSON-RPC framing, local subprocess lifecycle.
3. **ACP default command mapping matches provider families**
- Type: unit
- Disposition: new
- Harness: `fabro-acp` command mapping tests.
- Preconditions: `fabro-acp` crate exists with `agent-client-protocol-tokio = 0.11.1`.
- Actions: call `default_acp_command` for Anthropic, OpenAI, Kimi, Zai, Minimax, Inception, OpenAI-compatible, and Gemini.
- Expected outcome: Anthropic maps to `npx -y @zed-industries/claude-code-acp@latest`; OpenAI-compatible family maps to `npx -y @zed-industries/codex-acp@latest`; Gemini maps to `npx -y -- @google/gemini-cli@latest --experimental-acp`. Source of truth: implementation plan User-Visible Behavior default ACP command mapping.
- Interactions: provider enum coverage and ACP Tokio parser defaults.
4. **ACP command overrides are parsed as stdio commands, not raw shell**
- Type: boundary
- Disposition: new
- Harness: `fabro-acp` command parsing tests using `agent_client_protocol_tokio::AcpAgent::from_str`.
- Preconditions: no sandbox required.
- Actions: resolve `acp_command` values for a shell-word command, a blank string, a JSON stdio config with args/env, and a non-stdio JSON config.
- Expected outcome: shell-word and JSON stdio commands expose parsed program/args/env; blank overrides fail with `acp_command must not be empty`; HTTP/SSE configs fail with `only stdio ACP commands are supported`; rendered sandbox command uses parsed parts with shell quoting. Source of truth: implementation plan command override contract and shell quoting invariant in `AGENTS.md`.
- Interactions: ACP Tokio parser, command rendering, env merge inputs.
5. **Local sandbox stdio round-trips without a PTY**
- Type: integration
- Disposition: new
- Harness: `fabro-sandbox` local stdio process harness.
- Preconditions: temp local sandbox workspace; Python or a POSIX shell command available.
- Actions: spawn a line-oriented process with `spawn_stdio_process`, write `abc\n` to stdin, read one stdout line, then terminate and wait.
- Expected outcome: stdout returns the transformed line, stderr remains separately collectible, and `terminate()` completes without leaking the process. Source of truth: implementation plan Contracts And Invariants requiring sandbox-backed, bidirectional, non-PTY stdio.
- Interactions: local process groups, env filtering, async IO, cancellation cleanup.
6. **Sandbox providers preserve or reject ACP stdio capability correctly**
- Type: invariant
- Disposition: new
- Harness: `fabro-sandbox` provider/decorator tests.
- Preconditions: local sandbox, read/write guard, worktree/decorator wrappers, test-support sandbox, and Daytona provider stub are available.
- Actions: call `spawn_stdio_process` through each wrapper around a supporting sandbox; call it on Daytona; construct Docker exec stdio options.
- Expected outcome: wrappers forward to the inner sandbox; Daytona returns `ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet`; Docker create/start options attach stdin/stdout/stderr and set `tty=false`; Docker termination uses the stop-file/control path. Source of truth: implementation plan provider support and PTY corruption risk.
- Interactions: decorator macro, worktree path resolution, Docker option builder, Daytona provider boundary.
7. **ACP lifecycle initializes, creates a session, sends a prompt, and aggregates text**
- Type: integration
- Disposition: new
- Harness: `fabro-acp` fake ACP agent harness.
- Preconditions: fake ACP agent configured to emit two text `agent_message_chunk` updates and return `stopReason: "end_turn"`.
- Actions: call `run_acp_turn` with a prompt and cwd.
- Expected outcome: fake agent observes `initialize`, `session/new`, `session/prompt` in order; result text is the concatenation of text chunks; stop reason is `EndTurn`. Source of truth: ACP initialization/session/prompt docs and docs.rs quick-start lifecycle.
- Interactions: official ACP SDK client, sandbox stdio transport, JSON-RPC ordering.
8. **ACP runs inside the active sandbox and sees the workflow cwd**
- Type: integration
- Disposition: new
- Harness: `fabro-acp` fake agent plus local sandbox stdio.
- Preconditions: temp sandbox workspace; fake agent writes `hello.txt` in its cwd during `session/prompt`.
- Actions: call `run_acp_turn`, then inspect the sandbox workspace for `hello.txt`.
- Expected outcome: file exists inside the sandbox workspace, not the host process cwd; `session/new` cwd matches `sandbox.working_directory()`. Source of truth: implementation plan Contracts And Invariants requiring ACP processes to run inside the active Fabro sandbox.
- Interactions: sandbox cwd resolution, command launch, file mutation visibility.
9. **ACP permission requests auto-select an allow option**
- Type: integration
- Disposition: new
- Harness: `fabro-acp` fake ACP agent harness.
- Preconditions: fake agent sends `session/request_permission` with `AllowAlways`, `AllowOnce`, and reject options before completing the prompt.
- Actions: call `run_acp_turn` and record the client response.
- Expected outcome: client responds with the `AllowAlways` option id when present, then the turn continues and returns text. Source of truth: implementation plan permission handling contract; ACP supports agent-to-client permission requests.
- Interactions: ACP client request handler, cancellation token state, prompt turn progress.
10. **ACP cancellation sends session cancel and returns cancellation**
- Type: boundary
- Disposition: new
- Harness: `fabro-acp` fake ACP agent harness.
- Preconditions: fake agent has created a session and is holding `session/prompt` open.
- Actions: start `run_acp_turn`, cancel the token before completion, and let the fake agent record incoming notifications.
- Expected outcome: client sends `session/cancel` for the active session, terminates if the agent does not finish within grace, and returns `AcpError::Cancelled`. If a permission request arrives after cancellation, the response is `RequestPermissionOutcome::Cancelled`. Source of truth: implementation plan cancellation contract and ACP prompt lifecycle stop reasons.
- Interactions: cancel token, JSON-RPC notification, process termination.
11. **ACP timeout terminates the process and reports timeout**
- Type: boundary
- Disposition: new
- Harness: `fabro-acp` fake ACP agent harness.
- Preconditions: fake agent never responds to `session/prompt`; request timeout is short.
- Actions: call `run_acp_turn`.
- Expected outcome: process is terminated, stderr tail is available if emitted, and error is `AcpError::TimedOut`. Source of truth: implementation plan timeout contract using node timeout like CLI mode.
- Interactions: watchdog activity, process handle termination, stderr collector.
12. **ACP protocol failures include diagnostic stderr without losing typed errors**
- Type: boundary
- Disposition: new
- Harness: `fabro-acp` fake ACP agent harness.
- Preconditions: fake agents for malformed JSON-RPC and early nonzero exit.
- Actions: call `run_acp_turn` for each failure mode.
- Expected outcome: malformed JSON returns a protocol error; early exit includes exit status and stderr tail; error source chains remain inspectable where applicable. Source of truth: implementation plan malformed/early-exit behavior and error-handling strategy.
- Interactions: ACP SDK error propagation, stderr tail collection, process wait.
13. **ACP stop reasons map to Fabro backend outcomes**
- Type: boundary
- Disposition: new
- Harness: `fabro-acp` fake agent plus workflow `AgentAcpBackend` adapter tests.
- Preconditions: fake agent can return `EndTurn`, `Refusal`, `Cancelled`, `MaxTokens`, and `MaxTurnRequests`.
- Actions: run an ACP backend turn for each stop reason.
- Expected outcome: `EndTurn` and `Refusal` return text; `Cancelled` maps to `Error::Cancelled`; `MaxTokens` and `MaxTurnRequests` return handler errors containing the stop reason and partial output. Source of truth: implementation plan Stop reason handling.
- Interactions: protocol result mapping, workflow error conversion, event terminal paths.
14. **ACP backend adapter prepares credentials, env, Node runtime, and changed files**
- Type: integration
- Disposition: new
- Harness: `fabro-workflow` ACP adapter tests with fake credential resolver and fake sandbox.
- Preconditions: node uses `backend="acp"`; fake resolver can provide env vars and login command; sandbox records commands and git status before/after.
- Actions: call `AgentAcpBackend::run`.
- Expected outcome: login command runs before ACP; tool env overlays command env; default `npx` commands trigger Node/npm/npx bootstrap; explicit `acp_command` does not install provider CLIs; `files_touched` excludes pre-existing dirty files and includes new changed/untracked files. Source of truth: implementation plan env preparation, Node bootstrap, and changed-file semantics.
- Interactions: credential resolver, workflow tool env, sandbox exec, Git diff helper.
15. **ACP one-shot prompt nodes use sandboxed ACP and combine system prompt correctly**
- Type: integration
- Disposition: new
- Harness: `fabro-workflow` `PromptHandler` and `AgentAcpBackend::one_shot` tests.
- Preconditions: prompt node has `backend="acp"`; project memory can produce a system prompt; fake backend captures sandbox pointer and cancellation token.
- Actions: execute the prompt handler.
- Expected outcome: `PromptHandler` passes the active sandbox and run cancel token into `CodergenBackend::one_shot`; ACP one-shot sends `System:\n{system_prompt}\n\nUser:\n{prompt}` when system prompt exists and only the prompt when absent; no host process is used. Source of truth: implementation plan User-Visible Behavior for prompt/one_shot ACP support.
- Interactions: prompt handler, memory discovery, backend trait signature, run services.
16. **Backend router selects api, cli, and acp explicitly**
- Type: integration
- Disposition: extend
- Harness: `fabro-workflow` router tests.
- Preconditions: router has API, CLI, and ACP test backends with distinguishable responses.
- Actions: run agent nodes with absent backend, `backend="api"`, `backend="cli"`, `backend="acp"`, and `backend="codex"`.
- Expected outcome: absent and `api` use API; `cli` uses CLI; `acp` uses ACP; unknown backend fails with `unsupported LLM backend "codex"; expected one of: api, cli, acp`. Source of truth: implementation plan three-way router selection and strict validation requirement.
- Interactions: node attributes, model fallback, handler errors.
17. **Prompt router keeps legacy cli one-shot fallback but routes acp to ACP**
- Type: regression
- Disposition: extend
- Harness: `fabro-workflow` router one-shot tests.
- Preconditions: router has API and ACP one-shot test backends.
- Actions: call `one_shot` for prompt nodes with absent backend, `backend="api"`, `backend="cli"`, and `backend="acp"`.
- Expected outcome: absent, `api`, and legacy `cli` prompt nodes use API; `acp` uses ACP. Source of truth: implementation plan compatibility note for prompt nodes with `backend="cli"` and explicit ACP prompt support.
- Interactions: backend routing, prompt handler behavior, backward compatibility.
18. **Workflow validation accepts only supported backend values**
- Type: boundary
- Disposition: new
- Harness: `fabro-validate` `backend_valid` rule tests and CLI validate coverage if practical.
- Preconditions: graphs with absent backend and with `api`, `cli`, `acp`, and `codex`.
- Actions: run `fabro_validate::validate` against each graph; optionally run `fabro validate` against an invalid fixture.
- Expected outcome: absent, `api`, `cli`, and `acp` have no backend diagnostic; `codex` returns an error diagnostic containing `unsupported LLM backend "codex"; expected one of: api, cli, acp`. Source of truth: implementation plan User-Visible Behavior for unknown backend values.
- Interactions: validation registry, parser, CLI diagnostic rendering.
19. **Imported workflow placeholders propagate acp_command**
- Type: regression
- Disposition: extend
- Harness: `fabro-workflow` import transform tests.
- Preconditions: host workflow has an import placeholder with `backend="acp"` and `acp_command="python fake_agent.py"`; imported workflow has LLM nodes.
- Actions: run `ImportTransform` and inspect imported node attrs.
- Expected outcome: imported LLM nodes receive `backend="acp"` and the placeholder `acp_command`; unsupported placeholder attributes still poison the placeholder. Source of truth: implementation plan file list and import transform requirement.
- Interactions: graph transform, default attribute propagation, import validation.
20. **ACP events serialize with stage-scoped metadata**
- Type: integration
- Disposition: new
- Harness: `fabro-workflow` event conversion tests.
- Preconditions: construct `Event::AgentAcpStarted`, `AgentAcpCompleted`, `AgentAcpCancelled`, and `AgentAcpTimedOut` with a `StageScope`.
- Actions: convert each event through `to_run_event`.
- Expected outcome: event names are `agent.acp.started`, `agent.acp.completed`, `agent.acp.cancelled`, and `agent.acp.timed_out`; envelope includes `node_id`, stage id/visit-derived fields, and no prompt/env/credential contents. Source of truth: events strategy and implementation plan ACP event contract.
- Interactions: event naming, stored fields, `fabro-types` event body serde.
21. **Run projection records ACP provider metadata and terminal output**
- Type: integration
- Disposition: new
- Harness: `fabro-store` run projection tests.
- Preconditions: event sequence has stage start, `agent.acp.started`, terminal ACP event, and stage completion/failure.
- Actions: apply events to `RunProjection`.
- Expected outcome: `stage.provider_used.mode == "acp"` with provider, model, and command; completed output contains aggregated text/stderr payload; cancelled and timed-out terminal events set `CommandTermination::Cancelled` and `CommandTermination::TimedOut`. Source of truth: implementation plan run projection support.
- Interactions: stored event fields, stage lookup by visit, projection terminal data.
22. **Fork replay preserves ACP stage metadata**
- Type: regression
- Disposition: new
- Harness: `fabro-workflow` fork replay tests.
- Preconditions: source run history includes ACP started/cancelled/timed-out events before a checkpoint.
- Actions: call fork replay filtering or run a lower-level fork projection test.
- Expected outcome: `AgentAcpStarted`, `AgentAcpCancelled`, and `AgentAcpTimedOut` are replayed into the fork projection; `AgentAcpCompleted` follows the existing CLI completed replay policy. Source of truth: implementation plan fork replay requirement.
- Interactions: historical event filtering, forked run projection.
23. **ACP running stages are not steerable through the server API**
- Type: scenario
- Disposition: new
- Harness: server event-state harness extension.
- Preconditions: a running managed run with worker control channel; no active API-mode agent session; active stage marker has been set by `agent.acp.started`.
- Actions: call `POST /runs/{id}/steer` with a plain steer request and with interrupt+steer.
- Expected outcome: response is `409 CONFLICT` with a clear non-steerable-agent error code/message; no worker control message is enqueued as if an API session might appear. Source of truth: implementation plan server steerability tracking.
- Interactions: run manager event reducer, HTTP handler, worker control queue.
24. **ACP non-steerable marker clears on all terminal paths**
- Type: invariant
- Disposition: new
- Harness: server event-state harness extension.
- Preconditions: a running managed run with active ACP stage and no active API-mode stage.
- Actions: apply each clearing event independently: `agent.acp.completed`, `agent.acp.cancelled`, `agent.acp.timed_out`, `stage.completed`, and `stage.failed`; then call `POST /runs/{id}/steer` with a plain steer request.
- Expected outcome: plain steer is accepted/buffered after each terminal event because no non-steerable active agent remains. Source of truth: implementation plan server steerability clearing rules.
- Interactions: event reducer backstops, HTTP handler, stage lifecycle.
25. **Pipeline initialization wires ACP into real workflow handlers**
- Type: integration
- Disposition: new
- Harness: `fabro-workflow` pipeline initialization tests.
- Preconditions: graph contains `backend="acp"` LLM node; credentials are supplied through a stub/env source; dry-run and non-dry-run cases are both available.
- Actions: call `initialize`/`build_registry` and execute or resolve the node through the initialized registry using a fake ACP runner.
- Expected outcome: non-dry-run registry constructs a router with ACP; dry-run still builds no real backend and simulates LLM handlers; ACP does not fall back to host env when a resolver exists. Source of truth: implementation plan pipeline initialization task.
- Interactions: credential source, handler registry, dry-run path.
26. **Black-box `fabro run` executes an ACP-backed agent workflow**
- Type: scenario
- Disposition: new
- Harness: workflow ACP runner harness in `fabro-cli/tests/it/workflow`.
- Preconditions: temp workflow has an agent node with `backend="acp"`, `provider="openai"`, `model="fake-acp"`, and `acp_command` pointing to the checked-in fake ACP agent; local sandbox is used.
- Actions: run the workflow through the CLI test command, then read run state/events through existing workflow helpers.
- Expected outcome: run succeeds; stage response contains concatenated chunks; `hello.txt` is included in `files_touched`; run projection has `provider_used.mode == "acp"`; `agent.acp.started` and `agent.acp.completed` events are present. Source of truth: user request for first-class `backend="acp"` and implementation plan black-box workflow coverage.
- Interactions: CLI command, parser, validation, pipeline initialization, sandbox stdio, ACP protocol, run store.
27. **Black-box ACP prompt workflow uses ACP instead of API**
- Type: scenario
- Disposition: new
- Harness: workflow ACP runner harness.
- Preconditions: temp workflow has a prompt/one_shot node with `backend="acp"` and fake ACP command.
- Actions: run the workflow through the CLI test command and inspect stage response/events.
- Expected outcome: prompt node succeeds through ACP, response is fake ACP text, and `agent.acp.*` provider metadata appears; no API-mode `agent.session.activated` event is needed for the prompt. Source of truth: implementation plan User-Visible Behavior for `backend="acp"` on prompt/one_shot nodes.
- Interactions: prompt handler, one-shot routing, pipeline initialization, run projection.
28. **Documentation examples and backend references include ACP without stale CLI prompt claims**
- Type: regression
- Disposition: extend
- Harness: documentation grep plus existing docs build if normally run in CI.
- Preconditions: docs have been updated.
- Actions: run `rg -n "backend=.*cli|backend: cli|backend.*api|CLI backend|cli mode|ACP" docs/public lib/crates -g '*.md' -g '*.mdx'` and `cd apps/marketing && bun run build` only if the touched docs are built by that package.
- Expected outcome: docs mention valid backend values `api`, `cli`, `acp`; `cli` is described as legacy; ACP sandbox and Daytona limitations are documented; no stale claim remains that prompt nodes use CLI mode. Source of truth: implementation plan documentation task.
- Interactions: public docs, marketing/docs build pipeline.
29. **Final targeted ACP verification passes**
- Type: invariant
- Disposition: new
- Harness: repository test suites named by the implementation plan.
- Preconditions: all implementation tasks complete.
- Actions: run:
`ulimit -n 4096 && cargo nextest run -p fabro-acp --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-sandbox --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-workflow --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-validate --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-store --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-server --run-ignored all --no-fail-fast`;
`ulimit -n 4096 && cargo nextest run -p fabro-cli --run-ignored all --no-fail-fast`.
- Expected outcome: every suite passes without skipped tests or live provider credentials. Source of truth: accepted strategy final verification and implementation plan Task 10.
- Interactions: all changed crates and user-visible workflow/server surfaces.
30. **Workspace-wide build, formatting, and lint gates pass**
- Type: invariant
- Disposition: existing
- Harness: repository-wide Cargo/rustfmt/clippy commands.
- Preconditions: targeted tests pass.
- Actions: run `cargo build --workspace`, `ulimit -n 4096 && cargo nextest run --workspace --run-ignored all --no-fail-fast`, `cargo +nightly-2026-04-14 fmt --check --all`, and `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`.
- Expected outcome: build, workspace tests, formatting, and clippy all pass with zero skipped tests. Source of truth: repository `AGENTS.md` build/test commands and the no-skipped-tests final-run requirement.
- Interactions: full workspace dependency graph, feature flags, generated code boundaries.
## Coverage Summary
Covered action space:
- Workflow authoring: `backend` absent, `api`, `cli`, `acp`, invalid values, stylesheet/import propagation, `acp_command` shell-word and JSON stdio overrides.
- Execution surfaces: agent nodes, prompt/one_shot nodes, local sandbox ACP execution, default command selection, explicit override execution, credentials/env, Node bootstrap, changed-file reporting, cancellation, timeout, and stop reason handling.
- Protocol behavior: `initialize`, `session/new`, `session/prompt`, `session/update` text aggregation, permission requests, `session/cancel`, malformed JSON-RPC, and early process exit.
- Provider/sandbox boundaries: local stdio, Docker non-PTY stdio option/control behavior, Daytona unsupported error, decorator forwarding.
- Product-visible state: ACP events, run projection `provider_used.mode == "acp"`, terminal output/termination, fork replay, CLI workflow run state, and server steerability API behavior.
- Regression protection: existing CLI routing/CLI parsing tests, existing MCP stdio tests, dry-run initialization, repository build/fmt/clippy.
Explicit exclusions:
- Live Anthropic/OpenAI/Gemini ACP adapter calls are excluded; fake ACP agents provide deterministic coverage without paid credentials. Risk: vendor-specific adapter quirks may escape until optional/live tests are added.
- Full live Docker ACP workflow execution is not required unless the existing test environment already provides Docker. Unit-level Docker exec option/control tests cover the non-PTY and termination contract. Risk: daemon-specific stream behavior could still differ from Bollard option construction.
- Remote ACP transports are excluded because the implementation plan supports only stdio in this cutover. Risk: none for the agreed scope.
- ACP client filesystem and terminal capabilities are excluded because Fabro intentionally advertises none in this cutover. Risk: agents requiring those client APIs will fail as documented rather than silently using unsafe host capabilities.

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,292 @@
# Fabro MCP Server Test Plan
## Harness Requirements
The agreed testing strategy still holds after reading the implementation plan. The plan narrows the tool contract to five Devin-shaped run tools and requires the implementation to live in a new `fabro-mcp-server` crate, but it does not add paid APIs, live LLM calls, external infrastructure, or browser/UI behavior. The highest-value evidence remains a real `fabro mcp start` subprocess driven over stdio and backed by Fabro's real local test server/auth harness.
1. **Deterministic MCP stdio fixture**
- **Does:** constructs the exact command, environment, and cwd used to spawn `env!("CARGO_BIN_EXE_fabro") mcp start`.
- **Exposes:** `command: Vec<String>`, `env: HashMap<String, String>`, and `current_dir: PathBuf` usable by both `fabro_mcp::client::McpClient` and raw `std::process::Command` tests.
- **Complexity:** low. Add a narrow helper in `lib/crates/fabro-cli/tests/it/cmd/mcp.rs`; if needed, add `fabro_test::isolated_env(home_dir)` to mirror `apply_test_isolation`.
- **Tests depending on it:** 5, 6, 7, 8, 9, 10, 15, 16, 17.
2. **MCP tool-call assertion helpers**
- **Does:** calls a named MCP tool, asserts tool success or tool error, extracts `structured_content`, and verifies fallback text is concise rather than a JSON dump.
- **Exposes:** `call_tool_json(...)`, `call_tool_error_text(...)`, and normalization helpers for run IDs, timestamps, paths, event IDs, cursors, durations, and elapsed times.
- **Complexity:** low to medium. Keep it local to `cmd/mcp.rs` unless more than one test file needs it.
- **Tests depending on it:** 8, 9, 10, 11, 12, 13, 15, 16, 17.
3. **Real authenticated Fabro server fixture**
- **Does:** starts `RealAuthHarness::start_with_dev_token(...)`, seeds CLI dev-token auth into the test home, creates dry-run workflows through public CLI/MCP/API surfaces, and shuts down the server.
- **Exposes:** API target URL, persisted auth entry, HTTP client/server-visible state checks, and workflow fixture paths.
- **Complexity:** medium, mostly reuse existing `lib/crates/fabro-cli/tests/it/support/auth_harness.rs`.
- **Tests depending on it:** 8, 10, 11, 12, 13, 14, 17.
## Test Plan
1. **`fabro mcp` help exposes the MCP namespace**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness through existing `fabro_snapshot!`
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp --help`.
- **Expected outcome:** stdout snapshots a `Model Context Protocol server` namespace with `start`, `config`, and `init` subcommands; stderr is empty; exit status is 0. Source of truth: user request for `fabro mcp start`, `fabro mcp config`, `fabro mcp init <agent>`, and implementation plan CLI contract.
- **Interactions:** clap command tree, global CLI flags, snapshot filters.
2. **`fabro mcp start --help` documents stdio startup options**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness through `fabro_snapshot!`
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp start --help`.
- **Expected outcome:** stdout snapshots usage `fabro mcp start [OPTIONS]` with `--server <SERVER>` and `--storage-dir <DIR>`; stderr is empty; exit status is 0. Source of truth: implementation plan CLI contract.
- **Interactions:** clap flattening for `ServerConnectionArgs`.
3. **`fabro mcp config --help` documents config rendering options**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness through `fabro_snapshot!`
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp config --help`.
- **Expected outcome:** stdout snapshots usage and the same connection override flags as `start`; stderr is empty; exit status is 0. Source of truth: implementation plan CLI contract.
- **Interactions:** clap command help and global CLI flags.
4. **`fabro mcp init --help` documents supported agent selection**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness through `fabro_snapshot!`
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp init --help`.
- **Expected outcome:** stdout snapshots required `<AGENT>` with supported values `claude`, `cursor`, and `windsurf`; exit status is 0. Source of truth: user request and implementation plan supported-agent contract.
- **Interactions:** clap value enum rendering.
5. **`fabro mcp config` prints generic MCP client JSON**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness plus structured JSON parsing
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp config`; parse stdout as JSON.
- **Expected outcome:** stdout is valid JSON with `mcpServers.fabro.command == "fabro"` and `args == ["mcp", "start"]`; stderr is empty; exit status is 0. Source of truth: Daytona-shaped user request and implementation plan config JSON contract.
- **Interactions:** config rendering, stdout contract for a non-stdio command.
6. **`fabro mcp config` preserves connection flags in generated startup args**
- **Type:** integration
- **Disposition:** new
- **Harness:** output capture harness plus structured JSON parsing
- **Preconditions:** isolated `TestContext`; no auth or server required.
- **Actions:** run `fabro mcp config --server https://example.test/api/v1 --storage-dir /tmp/fabro-mcp-storage`; parse stdout as JSON.
- **Expected outcome:** JSON contains `args == ["mcp", "start", "--server", "https://example.test/api/v1", "--storage-dir", "/tmp/fabro-mcp-storage"]`; stderr is empty; exit status is 0. Source of truth: implementation plan examples for flag preservation.
- **Interactions:** CLI argument forwarding into MCP client config.
7. **`fabro mcp init <agent>` writes idempotent config without clobbering unrelated keys**
- **Type:** integration
- **Disposition:** new
- **Harness:** direct filesystem artifact assertion in isolated home
- **Preconditions:** isolated `TestContext`; pre-existing Cursor config with `mcpServers.other` and unrelated top-level key.
- **Actions:** run `fabro mcp init cursor --server https://example.test/api/v1` twice; read `~/.cursor/mcp.json`.
- **Expected outcome:** parsed JSON preserves unrelated keys and existing `mcpServers.other`, contains exactly one `mcpServers.fabro` entry with command `fabro` and expected args, and the second run does not duplicate or reorder into an invalid shape. Source of truth: implementation plan idempotent config merge contract.
- **Interactions:** filesystem directory creation, JSON merge/write, test home isolation.
8. **`fabro mcp init` writes each supported agent path**
- **Type:** integration
- **Disposition:** new
- **Harness:** direct filesystem artifact assertion in isolated home
- **Preconditions:** isolated `TestContext`; no existing Claude, Cursor, or Windsurf config.
- **Actions:** run `fabro mcp init claude`, `fabro mcp init cursor`, and `fabro mcp init windsurf` in separate contexts; read the platform-specific config file for each.
- **Expected outcome:** each config file exists at the path named by the implementation plan and contains `mcpServers.fabro` with `command: "fabro"` and `args: ["mcp", "start"]`. Source of truth: implementation plan agent path contract.
- **Interactions:** platform-specific path selection, filesystem writes.
9. **`fabro mcp init` rejects invalid existing config without overwrite**
- **Type:** boundary
- **Disposition:** new
- **Harness:** output capture and filesystem artifact assertion
- **Preconditions:** isolated `TestContext`; Cursor config file contains invalid JSON bytes.
- **Actions:** run `fabro mcp init cursor`; read the same file after failure.
- **Expected outcome:** command exits non-zero with a clear error that includes the config path; the file content is byte-for-byte unchanged. Source of truth: implementation plan invalid JSON failure contract and error-handling strategy.
- **Interactions:** JSON parsing, write avoidance on error, CLI error rendering.
10. **`fabro mcp start` initializes over stdio and lists the five run tools**
- **Type:** scenario
- **Disposition:** new
- **Harness:** interaction harness using deterministic MCP stdio fixture and `fabro_mcp::client::McpClient`
- **Preconditions:** isolated `TestContext`; no auth; no live Fabro server.
- **Actions:** spawn `fabro mcp start`; perform MCP `initialize`; call `tools/list`.
- **Expected outcome:** initialize succeeds without auth/server connectivity; `tools/list` returns exactly `fabro_run_create`, `fabro_run_search`, `fabro_run_interact`, `fabro_run_gather`, and `fabro_run_events`, each with an input schema. Source of truth: MCP lifecycle/tools spec as captured in the agreed strategy and implementation plan exact tool list.
- **Interactions:** `rmcp` stdio transport, existing `fabro-mcp` client crate, child process lifecycle.
11. **`fabro mcp start` reserves stdout for JSON-RPC only**
- **Type:** regression
- **Disposition:** new
- **Harness:** raw subprocess stdio harness
- **Preconditions:** isolated `TestContext`; no auth; no live Fabro server.
- **Actions:** spawn `fabro mcp start`; write a JSON-RPC `initialize` request to stdin; read the first stdout line.
- **Expected outcome:** first stdout line parses as JSON and has `jsonrpc: "2.0"`; no leading human log/help text appears on stdout; stderr may contain logs. Source of truth: MCP stdio transport contract and implementation plan stdout invariant.
- **Interactions:** CLI logging initialization, raw process pipes, JSON-RPC framing.
12. **MCP startup and tool discovery are fast without auth or server**
- **Type:** invariant
- **Disposition:** new
- **Harness:** interaction harness plus timing assertion
- **Preconditions:** isolated `TestContext`; no auth; no live Fabro server.
- **Actions:** measure elapsed time for spawning `fabro mcp start`, initializing, and calling `tools/list`.
- **Expected outcome:** operation completes under a generous smoke threshold, initially 2 seconds unless CI evidence requires a documented adjustment; all five tools are listed. Source of truth: agreed testing strategy performance smoke and implementation plan lazy API connection invariant.
- **Interactions:** process startup, `rmcp` initialization, tool schema generation.
13. **`fabro_run_create` creates and starts a real dry-run using persisted CLI auth**
- **Type:** scenario
- **Disposition:** new
- **Harness:** interaction harness plus real authenticated Fabro server fixture
- **Preconditions:** `RealAuthHarness::start_with_dev_token(...)`; dev-token auth seeded into isolated home for the harness target; checked-in `simple.fabro` fixture installed.
- **Actions:** spawn `fabro mcp start --server <target>`; call `fabro_run_create` with one run using `workflow`, `dry_run: true`, `auto_approve: true`, and label `source=mcp-test`.
- **Expected outcome:** tool result is not an MCP error; `structured_content.runs[0]` includes a run id, workflow, `started: true`, and status; fallback text exists and does not start with `{` or `[`; server-visible state contains the created run. Source of truth: user request for run-management MCP tools, implementation plan create semantics, OpenAPI `POST /api/v1/runs`, and `POST /api/v1/runs/{id}/start`.
- **Interactions:** persisted CLI auth store, Fabro API client, manifest builder/validation, run engine dry-run path.
14. **`fabro_run_search` filters, paginates, and includes archived runs by default**
- **Type:** integration
- **Disposition:** new
- **Harness:** interaction harness plus real authenticated Fabro server fixture
- **Preconditions:** authenticated MCP server; at least two MCP-created dry-run runs with distinct labels; one terminal run archived through API or MCP.
- **Actions:** call `fabro_run_search` with `run_ids`, `workflow`, `labels`, `status`, `archived`, `first`, and `after` combinations.
- **Expected outcome:** results are normalized run summaries; filters include only matching runs; `first` limits page size and returns an opaque cursor when more results exist; archived runs appear unless `archived: false` is supplied. Source of truth: implementation plan search semantics and OpenAPI list-runs include-archived behavior adapted by the plan.
- **Interactions:** server run listing, status string normalization, timestamp/date parsing, cursor handling.
15. **`fabro_run_interact get/start/message/cancel` uses selector resolution and server APIs**
- **Type:** integration
- **Disposition:** new
- **Harness:** interaction harness plus mocked HTTP server for precise API call assertions
- **Preconditions:** isolated `TestContext`; HTTP mock server with `/api/v1/runs/resolve`, `/runs/{id}`, `/state`, `/start`, `/steer`, and `/cancel` endpoints; CLI auth seeded if the mock requires auth.
- **Actions:** call `fabro_run_interact` with actions `get`, `start`, `message` with `interrupt: true`, and `cancel`, using a workflow-name selector rather than the exact run id.
- **Expected outcome:** each action first resolves the selector through `/runs/resolve`; calls the matching endpoint; returns a structured object with `run_id`, `action`, and action-specific `result`; tool errors are not produced for mocked successful API responses. Source of truth: implementation plan interact semantics and OpenAPI operation descriptions for retrieve, state, start, steer, and cancel.
- **Interactions:** run selector semantics, API error conversion, structured content projection.
16. **`fabro_run_interact archive/unarchive` changes real server-visible archived state**
- **Type:** scenario
- **Disposition:** new
- **Harness:** interaction harness plus real authenticated Fabro server fixture
- **Preconditions:** authenticated MCP server; completed dry-run created through MCP or public CLI.
- **Actions:** call `fabro_run_interact` with `archive`; call `fabro_run_search` with `archived: true`; call `fabro_run_interact` with `unarchive`; call `fabro_run_search` with `archived: false`.
- **Expected outcome:** archive action succeeds for the terminal run; archived search shows the run; unarchive action succeeds; unarchived search shows the run as terminal and not archived. Source of truth: implementation plan interact actions and OpenAPI archive/unarchive contracts.
- **Interactions:** archive state transitions, list/search visibility, server-side idempotence.
17. **`fabro_run_interact get_questions/answer` maps answer JSON to the API contract**
- **Type:** integration
- **Disposition:** new
- **Harness:** interaction harness plus mocked HTTP server for endpoint/body assertions
- **Preconditions:** isolated `TestContext`; HTTP mock server returns pending questions and accepts answer submissions.
- **Actions:** call `fabro_run_interact` with `get_questions`; call `answer` using representative payloads: `true`, `false`, string text, `{ "option": "a" }`, `{ "options": ["a", "b"] }`, and `{ "text": "hello" }`.
- **Expected outcome:** `get_questions` returns the API question list projection; `answer` sends `SubmitAnswerRequest` wire shapes with `kind: yes`, `no`, `text`, `selected`, and `multi_selected`, and returns a successful structured action result. Source of truth: implementation plan answer mapping and `lib/crates/fabro-api/tests/submit_answer_request_round_trip.rs`.
- **Interactions:** generated API type shape, JSON body serialization, human-in-the-loop endpoints.
18. **`fabro_run_gather` waits for terminal runs and returns current state on timeout**
- **Type:** scenario
- **Disposition:** new
- **Harness:** interaction harness plus real authenticated Fabro server fixture
- **Preconditions:** authenticated MCP server; one completed dry-run and one submitted/non-terminal run available.
- **Actions:** call `fabro_run_gather` on the completed run; call it on the non-terminal run with `timeout_seconds: 1` and `poll_interval_seconds: 5`.
- **Expected outcome:** completed run result has `timed_out: false` and terminal status; timeout case returns a successful structured result with `timed_out: true`, current run summary, and bounded elapsed wall time rather than an MCP/process error. Source of truth: implementation plan gather semantics and agreed performance/timeout strategy.
- **Interactions:** selector resolution, polling loop, server retrieve endpoint, terminal status classification.
19. **`fabro_run_events` lists, details, searches, filters, paginates, and truncates events**
- **Type:** integration
- **Disposition:** new
- **Harness:** interaction harness plus real authenticated Fabro server fixture
- **Preconditions:** authenticated MCP server; completed dry-run with stored events.
- **Actions:** call `fabro_run_events` with `action: "list"` and `first`; call `details` with returned event ids; call `search` with a known event-name substring; call filters for `event_types`, `categories`, `direction: "desc"`, `after`, `offset`, `limit`, and a small `max_content_length`.
- **Expected outcome:** returned events belong to the run; list ordering and pagination match requested parameters; details returns only requested event ids; search returns serialized events containing the query; category filtering uses event-name prefix; oversized serialized payloads are truncated with `truncated: true`; `next_cursor` is derived from the last returned sequence. Source of truth: implementation plan events semantics and OpenAPI `GET /api/v1/runs/{id}/events`.
- **Interactions:** event store pagination, event-name/category derivation, JSON serialization/truncation.
20. **Local validation errors happen before auth or network lookup and do not stop the server**
- **Type:** boundary
- **Disposition:** new
- **Harness:** interaction harness with `--server http://127.0.0.1:9` and no auth
- **Preconditions:** isolated `TestContext`; no auth entry; unreachable server URL.
- **Actions:** call `fabro_run_gather` with 51 run ids; call `tools/list`; call `fabro_run_interact` action `message` without `message`; call `tools/list` again.
- **Expected outcome:** each invalid tool call returns an MCP tool error mentioning the invalid field (`run_ids` or `message`); no auth guidance or connection error masks the local validation failure; subsequent `tools/list` succeeds. Source of truth: implementation plan validate-before-client invariant and MCP tool-error contract.
- **Interactions:** parameter validation, lazy client initialization, MCP service liveness after errors.
21. **Auth failures use existing Fabro login guidance and remain tool errors**
- **Type:** boundary
- **Disposition:** new
- **Harness:** interaction harness with protected real or mocked API target
- **Preconditions:** isolated `TestContext`; no saved auth for the target; server requires auth.
- **Actions:** spawn `fabro mcp start --server <protected-target>`; call a valid read tool such as `fabro_run_search`.
- **Expected outcome:** call returns an MCP tool error, not process exit; error text includes `Run \`fabro auth login\` to authenticate.`; subsequent `tools/list` still succeeds. Source of truth: user request for no separate MCP auth and implementation plan auth invariant.
- **Interactions:** auth store lookup, client connection, error classification/rendering.
22. **Invalid create inputs are rejected with field-specific tool errors**
- **Type:** boundary
- **Disposition:** new
- **Harness:** interaction harness with no auth and unreachable server
- **Preconditions:** isolated `TestContext`; no auth entry.
- **Actions:** call `fabro_run_create` with empty `runs`, with 51 runs, and with `inputs` containing a null value.
- **Expected outcome:** each call returns an MCP tool error naming the invalid field/key before any auth/server error; server remains alive for a subsequent `tools/list`. Source of truth: implementation plan create validation and JSON-to-TOML null rejection.
- **Interactions:** schema/validation layer, JSON-to-TOML conversion.
23. **Run tool successes always include structured content and concise text**
- **Type:** invariant
- **Disposition:** new
- **Harness:** MCP tool-call assertion helpers reused by scenario tests
- **Preconditions:** any successful calls from tests 13, 14, 16, 18, and 19.
- **Actions:** for each successful call, inspect `CallToolResult`.
- **Expected outcome:** `structured_content` is present; at least one text content item is present; text content is short and does not begin with `{` or `[`; `is_error` is absent or false. Source of truth: implementation plan successful tool-result invariant.
- **Interactions:** `rmcp::model::CallToolResult` construction and MCP client display fallback.
24. **Pure conversion helpers cover JSON-to-TOML and answer-request mapping**
- **Type:** unit
- **Disposition:** new
- **Harness:** `cargo nextest run -p fabro-mcp-server run_tools`
- **Preconditions:** none beyond crate compilation.
- **Actions:** call conversion helpers directly for strings, bools, integers, floats, arrays, objects, null input, and every supported answer payload shape.
- **Expected outcome:** JSON-compatible input values map to equivalent `toml::Value`; null returns an error naming the key; answer payloads serialize to `SubmitAnswerRequest` wire JSON with documented `kind` values; unsupported answer objects return a tool error. Source of truth: implementation plan conversion requirements and `fabro-api` submit-answer round-trip tests.
- **Interactions:** serde, generated API types, conversion error text.
25. **Existing MCP client crate behavior is not regressed**
- **Type:** regression
- **Disposition:** existing
- **Harness:** existing `fabro-mcp` crate tests
- **Preconditions:** repository builds with the new `fabro-mcp-server` crate added.
- **Actions:** run `cargo nextest run -p fabro-mcp`.
- **Expected outcome:** existing stdio client initialize/list/call tests pass. Source of truth: existing automated evidence and implementation plan decision to keep `fabro-mcp` as the external MCP client crate.
- **Interactions:** workspace dependency feature unification for `rmcp`, existing client transport behavior.
26. **Relevant existing CLI run/auth regressions still pass**
- **Type:** regression
- **Disposition:** existing
- **Harness:** existing `fabro-cli` integration tests
- **Preconditions:** implementation complete.
- **Actions:** run the existing tests matching `scenario::auth::auth_login_refresh_logout_flow`, `scenario::lifecycle::dry_run_create_start_attach_works_with_default_run_lookup`, and `cmd::ps::ps_explicit_local_tcp_target_uses_auth_store`; if names drift, list tests and run the corresponding auth/lifecycle/local-target checks.
- **Expected outcome:** all selected tests pass unchanged. Source of truth: agreed strategy existing automated evidence and user requirement that MCP reuse CLI auth/config behavior.
- **Interactions:** auth refresh/logout, local server run lifecycle, server target resolution.
27. **Final MCP command contract and workspace checks pass**
- **Type:** regression
- **Disposition:** extend
- **Harness:** repository command checks
- **Preconditions:** all feature implementation and snapshots complete.
- **Actions:** run `cargo nextest run -p fabro-cli --test it cmd::mcp`, `cargo +nightly-2026-04-14 fmt --check --all`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`, `ulimit -n 4096 && cargo nextest run --workspace`, and `cargo insta pending-snapshots`.
- **Expected outcome:** MCP command tests pass; formatting and clippy pass; workspace tests pass; no pending snapshots remain unless explicitly inspected and accepted for this feature. Source of truth: repository `AGENTS.md` build/test commands and snapshot policy.
- **Interactions:** entire workspace, rustfmt/clippy pinned nightly, nextest parallelism and file descriptor limit.
## Coverage Summary
Covered action space:
- CLI executable commands: `fabro mcp --help`, `fabro mcp start --help`, `fabro mcp config --help`, `fabro mcp init --help`, `fabro mcp config`, `fabro mcp config --server --storage-dir`, and `fabro mcp init claude|cursor|windsurf`.
- MCP protocol actions: stdio process startup, `initialize`, `tools/list`, and `tools/call`.
- MCP tool actions: `fabro_run_create`; `fabro_run_search`; `fabro_run_interact` actions `get`, `start`, `message`, `cancel`, `archive`, `unarchive`, `get_questions`, `answer`; `fabro_run_gather`; `fabro_run_events` actions `list`, `details`, and `search`.
- Error and boundary behavior: invalid local parameters, too many run ids, null input conversion, missing action fields, unsupported answer shapes, invalid agent config JSON, missing auth, unreachable server after local validation, timeout expiry, and service liveness after tool errors.
- Integration boundaries: CLI auth store reuse, Fabro API client, real local Fabro server, run manifest construction/validation, event store, generated API answer types, and existing `fabro-mcp` client crate.
- Performance smoke: initialize plus `tools/list` without auth/server.
Explicitly excluded per the agreed strategy:
- Live LLM/provider tests. Dry-run workflows and local/mocked servers cover run-management behavior without external credentials or spend.
- Manual QA of agent apps. `init` tests assert Fabro's written config path and JSON merge contract, not whether Claude/Cursor/Windsurf accept the file in a live app.
- Browser/UI tests. This feature adds CLI and MCP stdio surfaces only.
- Differential tests against Daytona or Devin. Their docs inspired shape, but no runnable reference implementation is available or required.
Residual risks:
- Agent config formats may evolve externally; tests protect Fabro's chosen file/path contract only.
- MCP SDK behavior can change with `rmcp` upgrades; protocol tests and existing `fabro-mcp` tests should catch startup/list/call regressions.
- Full workspace tests may be slower and subject to local FD limits; use the documented `ulimit -n 4096` command before `cargo nextest run --workspace`.

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,762 @@
# Sandbox Real-Agent Smoke QA Plan
## Purpose
Manually smoke test the `add-acp-backend` branch with real LLM-backed agents across the full sandbox/agent matrix:
| Sandbox provider | Claude | Codex | Gemini |
| --- | --- | --- | --- |
| Local | Required | Required | Required |
| Docker | Required | Required | Required |
| Daytona | Required | Required | Required |
The key branch constraint is intentional: ACP requires bidirectional raw stdio and is supported by local and Docker in this cutover, but not by Daytona. This QA plan proves ACP works through local and Docker sandboxes for Claude, Codex, and Gemini; uses Daytona CLI execution for the same three agents as the positive Daytona coverage; and finally verifies ACP-on-Daytona fails clearly instead of falling back to host execution or PTY transport.
## Scope
In scope:
- A real Claude ACP-backed agent running through the local sandbox provider with no container.
- A real Codex ACP-backed agent running through the local sandbox provider with no container.
- A real Gemini ACP-backed agent running through the local sandbox provider with no container.
- A real Claude ACP-backed agent running in a Docker sandbox.
- A real Codex ACP-backed agent running in a Docker sandbox.
- A real Gemini ACP-backed agent running in a Docker sandbox.
- A real Claude CLI-backed agent running in a Daytona sandbox.
- A real Codex CLI-backed agent running in a Daytona sandbox.
- A real Gemini CLI-backed agent running in a Daytona sandbox.
- Optional Daytona API backend control coverage after the required 3x3 matrix.
- An ACP-backed node on Daytona returning the expected unsupported-provider failure.
- Evidence capture through `inspect`, `events`, `dump`, and optional preserved-sandbox SSH.
Out of scope:
- Automated nextest coverage.
- ACP positive execution on Daytona.
- Full regression of Docker or local ACP behavior beyond this real-agent smoke.
- Snapshot creation performance tuning beyond what is needed to run the smoke.
## Preconditions
- Current branch is `add-acp-backend`.
- Local host has Node/npx available for the no-container ACP smoke.
- Docker is available for the Docker sandbox smoke.
- Daytona API key is available with sandbox/snapshot scopes.
- Real LLM credentials are available for all three required agents.
- GitHub access is configured if the operator chooses not to use `skip_clone = true`.
- Network access from the host, Docker container, and Daytona sandbox allows installing CLI packages.
Recommended environment:
```bash
cargo build -p fabro-cli
export FABRO=./target/debug/fabro
set -a
source .env
set +a
$FABRO doctor -v
```
Required environment variables for the full matrix:
- `DAYTONA_API_KEY`
- `ANTHROPIC_API_KEY`
- `OPENAI_API_KEY`
- `GEMINI_API_KEY`
## Test Data Setup
Create a scratch directory for manual smoke files:
```bash
mkdir -p smoke tmp
```
Docker and Daytona smoke configs use `skip_clone = true` to avoid depending on pushed branch state. This keeps the test focused on runtime behavior and real agent execution. The local smoke intentionally runs directly in the current working tree because `provider = "local"` has no container boundary; remove `smoke_local_acp_*_result.txt`, `.fabro-smoke-*-acp`, and `.fabro-smoke-home/` during cleanup if they remain.
Agent definitions for the required matrix:
| Agent | Provider | Model | Credential | ACP command | Daytona CLI install |
| --- | --- | --- | --- | --- | --- |
| Claude | `anthropic` | `claude-haiku-4-5` | `ANTHROPIC_API_KEY` | `npx -y @zed-industries/claude-code-acp@latest` | `npm install -g @anthropic-ai/claude-code`; binary: `claude` |
| Codex | `openai` | `gpt-5.3-codex` | `OPENAI_API_KEY` | `npx -y @zed-industries/codex-acp@latest` | `npm install -g @openai/codex`; binary: `codex` |
| Gemini | `gemini` | `gemini-3.1-pro-preview` | `GEMINI_API_KEY` | `npx -y -- @google/gemini-cli@latest --experimental-acp` | `npm install -g @google/gemini-cli`; binary: `gemini` |
## Smoke 1: Local ACP Backend Matrix
### Goal
Prove real Claude, Codex, and Gemini ACP-backed agents can run through the local sandbox provider without a container, use bidirectional raw stdio, and mutate the local workflow filesystem.
Required local combinations:
| Agent | Workflow | Result file | Expected content |
| --- | --- | --- | --- |
| Claude | `smoke/local_acp_claude.toml` | `smoke_local_acp_claude_result.txt` | `local-acp-claude-ok` |
| Codex | `smoke/local_acp_codex.toml` | `smoke_local_acp_codex_result.txt` | `local-acp-codex-ok` |
| Gemini | `smoke/local_acp_gemini.toml` | `smoke_local_acp_gemini_result.txt` | `local-acp-gemini-ok` |
### Files
Create one graph per agent. The Claude graph is:
```dot
digraph LocalAcpClaudeSmoke {
graph [goal="Local ACP Claude backend smoke"]
start [shape=Mdiamond]
setup [shape=parallelogram, script="rm -f smoke_local_acp_claude_result.txt"]
work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="/bin/bash .fabro-smoke-claude-acp", prompt="Create a file named smoke_local_acp_claude_result.txt containing exactly: local-acp-claude-ok"]
verify [shape=parallelogram, script="test \"$(cat smoke_local_acp_claude_result.txt)\" = \"local-acp-claude-ok\" && cat smoke_local_acp_claude_result.txt"]
exit [shape=Msquare]
start -> setup -> work -> verify -> exit
}
```
Create matching Codex and Gemini graphs with these substitutions:
| Agent | Graph file | Provider | Model | ACP wrapper | Result file | Expected content |
| --- | --- | --- | --- | --- | --- | --- |
| Codex | `smoke/local_acp_codex.fabro` | `openai` | `gpt-5.3-codex` | `.fabro-smoke-codex-acp` | `smoke_local_acp_codex_result.txt` | `local-acp-codex-ok` |
| Gemini | `smoke/local_acp_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `.fabro-smoke-gemini-acp` | `smoke_local_acp_gemini_result.txt` | `local-acp-gemini-ok` |
Create `smoke/local_acp_claude.toml`:
```toml
_version = 1
[workflow]
graph = "local_acp_claude.fabro"
[run.sandbox]
provider = "local"
[[run.prepare.steps]]
script = '''
set -eu
NODE_DIR="$(dirname "$(command -v node)")"
NPX_PATH="$(command -v npx)"
cat > .fabro-smoke-claude-acp <<SH
set -eu
export HOME="\${HOME:-\$PWD/.fabro-smoke-home}"
mkdir -p "\$HOME"
export PATH="$NODE_DIR:/usr/local/bin:/usr/bin:/bin:\${PATH:-}"
exec "$NPX_PATH" -y @zed-industries/claude-code-acp@latest
SH
chmod +x .fabro-smoke-claude-acp
'''
```
Create matching Codex and Gemini TOML files:
- `smoke/local_acp_codex.toml` uses `graph = "local_acp_codex.fabro"`, writes `.fabro-smoke-codex-acp`, and the wrapper ends with `exec "$NPX_PATH" -y @zed-industries/codex-acp@latest`.
- `smoke/local_acp_gemini.toml` uses `graph = "local_acp_gemini.fabro"`, writes `.fabro-smoke-gemini-acp`, and the wrapper ends with `exec "$NPX_PATH" -y -- @google/gemini-cli@latest --experimental-acp`.
### Run
```bash
$FABRO run --auto-approve smoke/local_acp_claude.toml
$FABRO run --auto-approve smoke/local_acp_codex.toml
$FABRO run --auto-approve smoke/local_acp_gemini.toml
```
### Pass Criteria
- Each of the three local runs exits successfully.
- Each `verify` stage prints its expected `local-acp-<agent>-ok` content.
- `fabro events <run-id> --tail 200` for each run includes:
- `agent.acp.started`
- `agent.acp.completed`
- `stage.completed` for `work`
- `stage.completed` for `verify`
- Events for each run do not include `agent.session.activated` for the `work` stage.
- Events for each run do not include `agent.cli.started` for the `work` stage.
- `fabro inspect <run-id>` shows each run succeeded.
- The local working tree contains each expected `smoke_local_acp_<agent>_result.txt` file with exactly the expected content.
### Failure Notes
- Missing `node` or `npx` on the host is a setup failure for this no-container smoke.
- A successful run with API or CLI events instead of ACP events is a branch failure because ACP silently fell back to another backend.
- The local provider writes directly into the current working tree; check `git status` before cleanup.
## Smoke 2: Docker ACP Backend Matrix
### Goal
Prove real Claude, Codex, and Gemini ACP-backed agents can run inside Docker sandboxes, use bidirectional non-PTY stdio through Docker exec, and mutate only the container workspace.
Required Docker combinations:
| Agent | Workflow | Result file | Expected content |
| --- | --- | --- | --- |
| Claude | `smoke/docker_acp_claude.toml` | `smoke_docker_acp_claude_result.txt` | `docker-acp-claude-ok` |
| Codex | `smoke/docker_acp_codex.toml` | `smoke_docker_acp_codex_result.txt` | `docker-acp-codex-ok` |
| Gemini | `smoke/docker_acp_gemini.toml` | `smoke_docker_acp_gemini_result.txt` | `docker-acp-gemini-ok` |
### Files
Create one graph per agent. The Claude graph is:
```dot
digraph DockerAcpClaudeSmoke {
graph [goal="Docker ACP Claude backend smoke"]
start [shape=Mdiamond]
setup [shape=parallelogram, script="rm -f smoke_docker_acp_claude_result.txt"]
work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="/bin/bash .fabro-smoke-claude-acp", prompt="Create a file named smoke_docker_acp_claude_result.txt containing exactly: docker-acp-claude-ok"]
verify [shape=parallelogram, script="test \"$(cat smoke_docker_acp_claude_result.txt)\" = \"docker-acp-claude-ok\" && cat smoke_docker_acp_claude_result.txt"]
exit [shape=Msquare]
start -> setup -> work -> verify -> exit
}
```
Create matching Codex and Gemini graphs with these substitutions:
| Agent | Graph file | Provider | Model | ACP wrapper | Result file | Expected content |
| --- | --- | --- | --- | --- | --- | --- |
| Codex | `smoke/docker_acp_codex.fabro` | `openai` | `gpt-5.3-codex` | `.fabro-smoke-codex-acp` | `smoke_docker_acp_codex_result.txt` | `docker-acp-codex-ok` |
| Gemini | `smoke/docker_acp_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `.fabro-smoke-gemini-acp` | `smoke_docker_acp_gemini_result.txt` | `docker-acp-gemini-ok` |
Create `smoke/docker_acp_claude.toml`:
```toml
_version = 1
[workflow]
graph = "docker_acp_claude.fabro"
[run.sandbox]
provider = "docker"
preserve = true
[run.sandbox.docker]
image = "buildpack-deps:noble"
network_mode = "bridge"
memory_limit = "4GB"
cpu_quota = 200000
skip_clone = true
[[run.prepare.steps]]
script = '''
set -eu
mkdir -p "$HOME/.local"
if ! command -v node >/dev/null 2>&1; then
curl -fsSL https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.gz | tar -xz --strip-components=1 -C "$HOME/.local"
fi
export PATH="$HOME/.local/bin:$PATH"
npm --version
npx --version
NODE_DIR="$(dirname "$(command -v node)")"
NPX_PATH="$(command -v npx)"
cat > .fabro-smoke-claude-acp <<SH
set -eu
export HOME="\${HOME:-\$PWD/.fabro-smoke-home}"
mkdir -p "\$HOME"
export PATH="$NODE_DIR:/usr/local/bin:/usr/bin:/bin:\${PATH:-}"
exec "$NPX_PATH" -y @zed-industries/claude-code-acp@latest
SH
chmod +x .fabro-smoke-claude-acp
'''
```
Create matching Codex and Gemini TOML files:
- `smoke/docker_acp_codex.toml` uses `graph = "docker_acp_codex.fabro"`, writes `.fabro-smoke-codex-acp`, and the wrapper ends with `exec "$NPX_PATH" -y @zed-industries/codex-acp@latest`.
- `smoke/docker_acp_gemini.toml` uses `graph = "docker_acp_gemini.fabro"`, writes `.fabro-smoke-gemini-acp`, and the wrapper ends with `exec "$NPX_PATH" -y -- @google/gemini-cli@latest --experimental-acp`.
### Run
```bash
$FABRO run --auto-approve smoke/docker_acp_claude.toml
$FABRO run --auto-approve smoke/docker_acp_codex.toml
$FABRO run --auto-approve smoke/docker_acp_gemini.toml
```
### Pass Criteria
- Each of the three Docker runs exits successfully.
- Each `verify` stage prints its expected `docker-acp-<agent>-ok` content.
- `fabro events <run-id> --tail 200` for each run includes:
- `sandbox.ready`
- `setup.started`
- `setup.completed`
- `agent.acp.started`
- `agent.acp.completed`
- `stage.completed` for `verify`
- Events for each run do not include `agent.session.activated` for the `work` stage.
- Events for each run do not include `agent.cli.started` for the `work` stage.
- `fabro inspect <run-id>` shows each run succeeded.
- Each preserved Docker sandbox contains its expected `smoke_docker_acp_<agent>_result.txt` file with exactly the expected content.
### Failure Notes
- Docker daemon, image pull, or package-install failures are setup failures unless the error indicates ACP stdio or sandbox routing broke.
- A successful run with API or CLI events instead of ACP events is a branch failure because ACP silently fell back to another backend.
## Smoke 3: Daytona API Backend Control
### Goal
Optionally prove a real provider API agent can use Fabro-managed tools inside the Daytona sandbox and mutate the sandbox filesystem. This is a backend control smoke, not part of the required 3x3 external-agent matrix.
### Files
Create `smoke/daytona_api.fabro`:
```dot
digraph DaytonaApiSmoke {
graph [goal="Daytona API backend smoke"]
start [shape=Mdiamond]
setup [shape=parallelogram, script="rm -f smoke_api_result.txt"]
work [type="agent", backend="api", provider="anthropic", model="claude-haiku-4-5", prompt="Create a file named smoke_api_result.txt containing exactly: daytona-api-ok"]
verify [shape=parallelogram, script="test \"$(cat smoke_api_result.txt)\" = \"daytona-api-ok\" && cat smoke_api_result.txt"]
exit [shape=Msquare]
start -> setup -> work -> verify -> exit
}
```
Create `smoke/daytona_api.toml`:
```toml
_version = 1
[workflow]
graph = "daytona_api.fabro"
[run.sandbox]
provider = "daytona"
preserve = true
[run.sandbox.daytona]
skip_clone = true
auto_stop_interval = 60
```
### Run
```bash
$FABRO run --auto-approve smoke/daytona_api.toml
```
### Pass Criteria
- Run exits successfully.
- The `verify` stage prints `daytona-api-ok`.
- `fabro events <run-id> --tail 200` includes:
- `sandbox.ready`
- `agent.session.activated`
- `stage.completed` for `work`
- `stage.completed` for `verify`
- `fabro inspect <run-id>` shows the run succeeded.
### Failure Notes
- Provider authentication failures are setup failures unless the error indicates sandbox routing or missing Daytona state.
- Missing `smoke_api_result.txt` after a successful agent stage is a failure.
## Smoke 4: Daytona CLI Backend Matrix
### Goal
Prove the branch runs real Claude, Codex, and Gemini external CLI agents inside Daytona when the CLI is preinstalled by the workflow environment. This also confirms Fabro no longer installs CLIs implicitly at stage runtime.
Required Daytona combinations:
| Agent | Workflow | Result file | Expected content | CLI binary |
| --- | --- | --- | --- | --- |
| Claude | `smoke/daytona_cli_claude.toml` | `smoke_daytona_cli_claude_result.txt` | `daytona-cli-claude-ok` | `claude` |
| Codex | `smoke/daytona_cli_codex.toml` | `smoke_daytona_cli_codex_result.txt` | `daytona-cli-codex-ok` | `codex` |
| Gemini | `smoke/daytona_cli_gemini.toml` | `smoke_daytona_cli_gemini_result.txt` | `daytona-cli-gemini-ok` | `gemini` |
### Files
Create one graph per agent. The Claude graph is:
```dot
digraph DaytonaCliClaudeSmoke {
graph [goal="Daytona CLI Claude backend smoke"]
start [shape=Mdiamond]
setup [shape=parallelogram, script="rm -f smoke_daytona_cli_claude_result.txt"]
work [type="agent", backend="cli", provider="anthropic", model="claude-haiku-4-5", prompt="Create a file named smoke_daytona_cli_claude_result.txt containing exactly: daytona-cli-claude-ok"]
verify [shape=parallelogram, script="test \"$(cat smoke_daytona_cli_claude_result.txt)\" = \"daytona-cli-claude-ok\" && cat smoke_daytona_cli_claude_result.txt"]
exit [shape=Msquare]
start -> setup -> work -> verify -> exit
}
```
Create matching Codex and Gemini graphs with these substitutions:
| Agent | Graph file | Provider | Model | Result file | Expected content |
| --- | --- | --- | --- | --- | --- |
| Codex | `smoke/daytona_cli_codex.fabro` | `openai` | `gpt-5.3-codex` | `smoke_daytona_cli_codex_result.txt` | `daytona-cli-codex-ok` |
| Gemini | `smoke/daytona_cli_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `smoke_daytona_cli_gemini_result.txt` | `daytona-cli-gemini-ok` |
Create `smoke/daytona_cli_claude.toml`:
```toml
_version = 1
[workflow]
graph = "daytona_cli_claude.fabro"
[run.sandbox]
provider = "daytona"
preserve = true
[run.sandbox.daytona]
skip_clone = true
auto_stop_interval = 60
[[run.prepare.steps]]
script = '''
set -eu
mkdir -p "$HOME/.local"
if ! command -v node >/dev/null 2>&1; then
curl -fsSL https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.gz | tar -xz --strip-components=1 -C "$HOME/.local"
fi
export PATH="$HOME/.local/bin:$PATH"
npm config set prefix "$HOME/.local"
command -v claude >/dev/null 2>&1 || npm install -g @anthropic-ai/claude-code
claude --version
'''
```
Create matching Codex and Gemini TOML files:
- `smoke/daytona_cli_codex.toml` uses `graph = "daytona_cli_codex.fabro"`, installs `@openai/codex` when `codex` is missing, and prints `codex --version`.
- `smoke/daytona_cli_gemini.toml` uses `graph = "daytona_cli_gemini.fabro"`, installs `@google/gemini-cli` when `gemini` is missing, and prints `gemini --version`.
### Run
```bash
$FABRO run --auto-approve smoke/daytona_cli_claude.toml
$FABRO run --auto-approve smoke/daytona_cli_codex.toml
$FABRO run --auto-approve smoke/daytona_cli_gemini.toml
```
### Pass Criteria
- Each of the three Daytona CLI runs exits successfully.
- Each `verify` stage prints its expected `daytona-cli-<agent>-ok` content.
- `fabro events <run-id> --tail 200` for each run includes:
- `setup.started`
- `setup.completed`
- `agent.cli.started`
- `agent.cli.completed`
- `stage.completed` for `verify`
- Events for each run do not include new `cli.ensure.started`, `cli.ensure.completed`, or `cli.ensure.failed` entries for this branch's runtime path.
- Each preserved Daytona sandbox contains its expected `smoke_daytona_cli_<agent>_result.txt` file with exactly the expected content.
### Failure Notes
- `CLI backend requires '<binary>' to be installed in the sandbox PATH` means the prepare step did not install the agent CLI where the backend expects it. Treat this as environment/setup failure unless the prepare logs prove `claude`, `codex`, or `gemini` was installed in `$HOME/.local/bin`.
- CLI package-install failures may be caused by Daytona network policy or npm registry availability.
## Smoke 5: Daytona ACP Backend Expected Unsupported Failure
### Goal
Prove ACP on Daytona fails explicitly because Daytona lacks bidirectional raw stdio support. This provider-boundary check is run once with Claude because the failure must happen before any agent-specific command is launched. This test guards against unsafe fallbacks such as running ACP on the host or over a PTY.
### Files
Create `smoke/daytona_acp_unsupported.fabro`:
```dot
digraph DaytonaAcpUnsupportedSmoke {
graph [goal="ACP Daytona unsupported smoke"]
start [shape=Mdiamond]
work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="npx -y @zed-industries/claude-code-acp@latest", prompt="Create smoke_acp_result.txt"]
exit [shape=Msquare]
start -> work -> exit
}
```
Create `smoke/daytona_acp_unsupported.toml`:
```toml
_version = 1
[workflow]
graph = "daytona_acp_unsupported.fabro"
[run.sandbox]
provider = "daytona"
preserve = true
[run.sandbox.daytona]
skip_clone = true
auto_stop_interval = 60
```
### Run
```bash
$FABRO run --auto-approve smoke/daytona_acp_unsupported.toml
```
### Pass Criteria
- Run fails.
- The failure text contains:
- `ACP backend requires bidirectional stdio`
- `Daytona sandbox provider does not support it yet`
- Events include `agent.acp.started`.
- Events do not include `agent.acp.completed`.
- The preserved Daytona sandbox does not contain `smoke_acp_result.txt`.
### Failure Notes
- If the run succeeds, that is a failure for this branch because ACP should not execute on Daytona.
- If the failure is about `acp_command` missing, the workflow file is wrong.
- If the failure is about `npx` missing before the Daytona unsupported error, inspect the code path: the smoke should prove the sandbox stdio provider boundary, not package availability.
## Required 3x3 Matrix Record
The plan is incomplete until all nine sandbox/agent combinations below have a run ID and evidence:
| Sandbox | Agent | Backend | Workflow | Run ID |
| --- | --- | --- | --- | --- |
| Local | Claude | ACP | `smoke/local_acp_claude.toml` | `________________` |
| Local | Codex | ACP | `smoke/local_acp_codex.toml` | `________________` |
| Local | Gemini | ACP | `smoke/local_acp_gemini.toml` | `________________` |
| Docker | Claude | ACP | `smoke/docker_acp_claude.toml` | `________________` |
| Docker | Codex | ACP | `smoke/docker_acp_codex.toml` | `________________` |
| Docker | Gemini | ACP | `smoke/docker_acp_gemini.toml` | `________________` |
| Daytona | Claude | CLI | `smoke/daytona_cli_claude.toml` | `________________` |
| Daytona | Codex | CLI | `smoke/daytona_cli_codex.toml` | `________________` |
| Daytona | Gemini | CLI | `smoke/daytona_cli_gemini.toml` | `________________` |
## Verification Checklist
Use this checklist as the operator-facing record for the smoke. Fill in run IDs and notes as each step completes.
### Preconditions
- [ ] Current branch is `add-acp-backend`.
- [ ] `cargo build -p fabro-cli` completed successfully.
- [ ] `FABRO=./target/debug/fabro` is exported for the shell running the smoke.
- [ ] `.env` is loaded.
- [ ] `$FABRO doctor -v` completed without a blocking environment error.
- [ ] Host `node` is available for the local ACP smoke.
- [ ] Host `npx` is available for the local ACP smoke.
- [ ] Docker is available for the Docker sandbox smoke.
- [ ] `DAYTONA_API_KEY` is present and has sandbox/snapshot scopes.
- [ ] `ANTHROPIC_API_KEY` is present for Claude smokes.
- [ ] `OPENAI_API_KEY` is present for Codex smokes.
- [ ] `GEMINI_API_KEY` is present for Gemini smokes.
- [ ] Host, Docker, and Daytona network paths can install CLI packages.
- [ ] `smoke/` and `tmp/` directories exist.
### Smoke 1: Local ACP Backend Matrix
- [ ] Created `smoke/local_acp_claude.fabro` and `smoke/local_acp_claude.toml`.
- [ ] Created `smoke/local_acp_codex.fabro` and `smoke/local_acp_codex.toml`.
- [ ] Created `smoke/local_acp_gemini.fabro` and `smoke/local_acp_gemini.toml`.
- [ ] All three local configs use `provider = "local"`.
- [ ] Prepare steps create `.fabro-smoke-claude-acp`, `.fabro-smoke-codex-acp`, and `.fabro-smoke-gemini-acp`.
- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_claude.toml`.
- [ ] Recorded local Claude ACP run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_codex.toml`.
- [ ] Recorded local Codex ACP run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_gemini.toml`.
- [ ] Recorded local Gemini ACP run ID: `________________`.
- [ ] Each local ACP run exited successfully.
- [ ] Verify stages printed `local-acp-claude-ok`, `local-acp-codex-ok`, and `local-acp-gemini-ok`.
- [ ] Each local run's events include `agent.acp.started`.
- [ ] Each local run's events include `agent.acp.completed`.
- [ ] Each local run's events include `stage.completed` for `work`.
- [ ] Each local run's events include `stage.completed` for `verify`.
- [ ] Each local run's events do not include `agent.session.activated` for the `work` stage.
- [ ] Each local run's events do not include `agent.cli.started` for the `work` stage.
- [ ] `fabro inspect <run-id>` shows each local run succeeded.
- [ ] Local working tree contains the three expected `smoke_local_acp_<agent>_result.txt` files with exact contents.
### Smoke 2: Docker ACP Backend Matrix
- [ ] Created `smoke/docker_acp_claude.fabro` and `smoke/docker_acp_claude.toml`.
- [ ] Created `smoke/docker_acp_codex.fabro` and `smoke/docker_acp_codex.toml`.
- [ ] Created `smoke/docker_acp_gemini.fabro` and `smoke/docker_acp_gemini.toml`.
- [ ] All three Docker configs use Docker with `preserve = true`.
- [ ] All three Docker configs use `skip_clone = true`.
- [ ] Prepare steps install or verify Node.
- [ ] Prepare steps verify `npx`.
- [ ] Prepare steps create `.fabro-smoke-claude-acp`, `.fabro-smoke-codex-acp`, and `.fabro-smoke-gemini-acp`.
- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_claude.toml`.
- [ ] Recorded Docker Claude ACP run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_codex.toml`.
- [ ] Recorded Docker Codex ACP run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_gemini.toml`.
- [ ] Recorded Docker Gemini ACP run ID: `________________`.
- [ ] Each Docker ACP run exited successfully.
- [ ] Verify stages printed `docker-acp-claude-ok`, `docker-acp-codex-ok`, and `docker-acp-gemini-ok`.
- [ ] Each Docker run's events include `sandbox.ready`.
- [ ] Each Docker run's events include `setup.started`.
- [ ] Each Docker run's events include `setup.completed`.
- [ ] Each Docker run's events include `agent.acp.started`.
- [ ] Each Docker run's events include `agent.acp.completed`.
- [ ] Each Docker run's events include `stage.completed` for `verify`.
- [ ] Each Docker run's events do not include `agent.session.activated` for the `work` stage.
- [ ] Each Docker run's events do not include `agent.cli.started` for the `work` stage.
- [ ] `fabro inspect <run-id>` shows each Docker run succeeded.
- [ ] Preserved Docker sandboxes contain their expected `smoke_docker_acp_<agent>_result.txt` files with exact contents.
### Smoke 3: Daytona API Backend Control
- [ ] Created `smoke/daytona_api.fabro`.
- [ ] Created `smoke/daytona_api.toml`.
- [ ] Config uses Daytona with `preserve = true`.
- [ ] Config uses `skip_clone = true`.
- [ ] Ran `$FABRO run --auto-approve smoke/daytona_api.toml`.
- [ ] Recorded Daytona API smoke run ID: `________________`.
- [ ] Run exited successfully.
- [ ] `verify` stage printed `daytona-api-ok`.
- [ ] Events include `sandbox.ready`.
- [ ] Events include `agent.session.activated`.
- [ ] Events include `stage.completed` for `work`.
- [ ] Events include `stage.completed` for `verify`.
- [ ] `fabro inspect <run-id>` shows the run succeeded.
- [ ] Preserved Daytona sandbox contains `smoke_api_result.txt` with exactly `daytona-api-ok`.
### Smoke 4: Daytona CLI Backend Matrix
- [ ] Created `smoke/daytona_cli_claude.fabro` and `smoke/daytona_cli_claude.toml`.
- [ ] Created `smoke/daytona_cli_codex.fabro` and `smoke/daytona_cli_codex.toml`.
- [ ] Created `smoke/daytona_cli_gemini.fabro` and `smoke/daytona_cli_gemini.toml`.
- [ ] All three Daytona CLI configs use Daytona with `preserve = true`.
- [ ] All three Daytona CLI configs use `skip_clone = true`.
- [ ] Prepare steps install or verify Node.
- [ ] Prepare steps install or verify `claude`, `codex`, and `gemini` in the sandbox PATH.
- [ ] Prepare steps print `claude --version`, `codex --version`, and `gemini --version`.
- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_claude.toml`.
- [ ] Recorded Daytona Claude CLI run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_codex.toml`.
- [ ] Recorded Daytona Codex CLI run ID: `________________`.
- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_gemini.toml`.
- [ ] Recorded Daytona Gemini CLI run ID: `________________`.
- [ ] Each Daytona CLI run exited successfully.
- [ ] Verify stages printed `daytona-cli-claude-ok`, `daytona-cli-codex-ok`, and `daytona-cli-gemini-ok`.
- [ ] Each Daytona CLI run's events include `setup.started`.
- [ ] Each Daytona CLI run's events include `setup.completed`.
- [ ] Each Daytona CLI run's events include `agent.cli.started`.
- [ ] Each Daytona CLI run's events include `agent.cli.completed`.
- [ ] Each Daytona CLI run's events include `stage.completed` for `verify`.
- [ ] Each Daytona CLI run's events do not include `cli.ensure.started`.
- [ ] Each Daytona CLI run's events do not include `cli.ensure.completed`.
- [ ] Each Daytona CLI run's events do not include `cli.ensure.failed`.
- [ ] Preserved Daytona sandboxes contain their expected `smoke_daytona_cli_<agent>_result.txt` files with exact contents.
### Smoke 5: Daytona ACP Unsupported Failure
- [ ] Created `smoke/daytona_acp_unsupported.fabro`.
- [ ] Created `smoke/daytona_acp_unsupported.toml`.
- [ ] Config uses Daytona with `preserve = true`.
- [ ] Config uses `skip_clone = true`.
- [ ] Ran `$FABRO run --auto-approve smoke/daytona_acp_unsupported.toml`.
- [ ] Recorded Daytona ACP smoke run ID: `________________`.
- [ ] Run failed.
- [ ] Failure text contains `ACP backend requires bidirectional stdio`.
- [ ] Failure text contains `Daytona sandbox provider does not support it yet`.
- [ ] Events include `agent.acp.started`.
- [ ] Events do not include `agent.acp.completed`.
- [ ] Preserved Daytona sandbox does not contain `smoke_acp_result.txt`.
- [ ] No evidence shows ACP ran on the host.
- [ ] No evidence shows ACP used a PTY fallback.
- [ ] No evidence shows ACP silently fell back to API or CLI.
### Evidence Capture
For each required run:
- [ ] Captured `$FABRO inspect <run-id>`.
- [ ] Captured `$FABRO events <run-id> --tail 200`.
- [ ] Captured `$FABRO dump --output tmp/<run-id>-dump <run-id>`.
- [ ] Recorded command used.
- [ ] Recorded final status.
- [ ] Recorded relevant event names.
- [ ] Recorded any external-provider or sandbox infrastructure errors.
For provider-specific filesystem checks:
- [ ] Inspected local working tree files directly.
- [ ] Inspected preserved Docker filesystem with `$FABRO sandbox ssh <run-id>`.
- [ ] Inspected preserved Daytona filesystem with `$FABRO sandbox ssh <run-id>`.
- [ ] Recorded whether expected files exist in the expected provider workspace.
### Cleanup And Final Acceptance
- [ ] Removed each run with `$FABRO rm -f <run-id>` after evidence capture.
- [ ] Removed local smoke artifacts: `smoke_local_acp_*_result.txt`, `.fabro-smoke-*-acp`, and `.fabro-smoke-home/`.
- [ ] Verified preserved Docker sandbox is gone with Docker or `fabro inspect <run-id>`.
- [ ] Verified preserved Daytona sandboxes are gone from the Daytona dashboard or `fabro inspect <run-id>`.
- [ ] Local ACP backend smoke succeeded with Claude, Codex, and Gemini and no API/CLI fallback.
- [ ] Docker ACP backend smoke succeeded with Claude, Codex, and Gemini and no API/CLI fallback.
- [ ] Daytona CLI backend smoke succeeded with Claude, Codex, and Gemini after explicit CLI installation in prepare steps.
- [ ] Optional Daytona API backend control result was recorded if run.
- [ ] Daytona ACP smoke failed with the expected unsupported bidirectional-stdio message.
- [ ] Captured events and dumps are sufficient to diagnose any failure without rerunning immediately.
## Evidence Capture Commands
For each run, capture:
```bash
$FABRO inspect <run-id>
$FABRO events <run-id> --tail 200
$FABRO dump --output tmp/<run-id>-dump <run-id>
```
For the local smoke, inspect the local filesystem. For preserved Docker and Daytona sandboxes, inspect the provider filesystem:
```bash
cat smoke_local_acp_claude_result.txt 2>/dev/null || true
cat smoke_local_acp_codex_result.txt 2>/dev/null || true
cat smoke_local_acp_gemini_result.txt 2>/dev/null || true
$FABRO sandbox ssh <run-id>
pwd
ls -la
cat smoke_docker_acp_claude_result.txt 2>/dev/null || true
cat smoke_docker_acp_codex_result.txt 2>/dev/null || true
cat smoke_docker_acp_gemini_result.txt 2>/dev/null || true
cat smoke_api_result.txt 2>/dev/null || true
cat smoke_daytona_cli_claude_result.txt 2>/dev/null || true
cat smoke_daytona_cli_codex_result.txt 2>/dev/null || true
cat smoke_daytona_cli_gemini_result.txt 2>/dev/null || true
cat smoke_acp_result.txt 2>/dev/null || true
exit
```
Record for each run:
- Run ID.
- Command used.
- Final status.
- Relevant event names.
- Whether expected files exist in the expected local, Docker, or Daytona workspace.
- Any external-provider or sandbox infrastructure errors.
## Cleanup
After evidence capture:
```bash
$FABRO rm -f <run-id>
rm -f smoke_local_acp_*_result.txt .fabro-smoke-*-acp
rm -rf .fabro-smoke-home
```
Verify preserved Docker and Daytona sandboxes are gone from Docker/Daytona or by rerunning `fabro inspect <run-id>` and confirming no active sandbox remains.
## Final Acceptance Criteria
The branch passes this manual QA plan when:
1. The local sandbox provider succeeds with real Claude, Codex, and Gemini ACP-backed agents.
2. The Docker sandbox provider succeeds with real Claude, Codex, and Gemini ACP-backed agents.
3. The Daytona sandbox provider succeeds with real Claude, Codex, and Gemini CLI-backed agents after explicit CLI installation in prepare steps.
4. The optional Daytona API control smoke, if run, succeeds or has a clearly recorded external-provider/setup failure.
5. The ACP Daytona smoke fails with the expected unsupported bidirectional-stdio message.
6. No evidence shows ACP-on-Daytona ran on the host, used a PTY fallback, or silently fell back to API/CLI.
7. Captured run events and dumps are sufficient to diagnose any failure without rerunning immediately.

View file

@ -1,11 +1,40 @@
---
title: "MCP"
description: "Extend agents with Model Context Protocol servers"
description: "Connect MCP tools to agents and expose Fabro runs to MCP clients"
---
MCP ([Model Context Protocol](https://modelcontextprotocol.io/)) lets you connect external tool servers to Fabro agents. An MCP server exposes tools over a standardized protocol — databases, APIs, file systems, custom services — and Fabro discovers and registers them automatically. Agents call MCP tools the same way they call built-in tools.
## How it works
Fabro can also run as an MCP server. MCP clients can use Fabro's run-management tools to create, inspect, control, wait for, and read events from workflow runs through the authenticated `fabro` CLI.
## Fabro as an MCP server
Use `fabro mcp init` to configure an MCP client to launch Fabro:
```bash
fabro mcp init claude
```
Supported client targets are `claude`, `cursor`, and `windsurf`. The generated configuration launches `fabro mcp start` over stdio and reuses the CLI's normal server selection, OAuth refresh, dev-token handling, proxy behavior, and local storage.
You can also print the MCP configuration JSON or start the server directly:
```bash
fabro mcp config
fabro mcp start
```
Pass `--server` when the MCP client should connect to a specific Fabro server, or `--storage-dir` when it should use a non-default CLI storage directory.
| Tool | Purpose |
|---|---|
| `fabro_run_create` | Create one or more workflow runs, starting them by default. |
| `fabro_run_search` | Search runs by ID, workflow, labels, status, archive state, and creation time. |
| `fabro_run_interact` | Get, start, message, cancel, archive, unarchive, inspect questions, or answer a run. |
| `fabro_run_gather` | Wait for runs to reach terminal states, returning current state on timeout. |
| `fabro_run_events` | List, inspect, or search stored events for a run. |
## Fabro agents as MCP clients
When an agent session starts with MCP servers configured, Fabro:
@ -26,12 +55,12 @@ mcp__{server}__{tool}
For example, a server named `filesystem` exposing a `read_file` tool becomes `mcp__filesystem__read_file`. Special characters in server or tool names (hyphens, dots, etc.) are replaced with underscores.
## Configuration
## Agent MCP configuration
MCP servers can be configured in two places:
MCP servers available to Fabro agents can be configured in two places:
- **`~/.fabro/settings.toml`** — applies to `fabro exec` sessions. See [User Configuration](/reference/user-configuration#mcp_servers-section).
- **Run config TOML** — applies to workflow runs (`fabro run`). See [Run Configuration](/execution/run-configuration#mcp_servers).
- **User configuration** — `~/.fabro/settings.toml` can define shared workflow MCP servers under `[run.agent.mcps.<name>]`, or `fabro exec`-only servers under `[cli.exec.agent.mcps.<name>]`. See [User Configuration](/reference/user-configuration).
- **Run config TOML** — workflow config can define run-specific MCP servers under `[run.agent.mcps.<name>]`. See [Run Configuration](/execution/run-configuration#runagentmcps).
Each server entry specifies a transport type and optional timeouts. The server name is the TOML table key and is used in qualified tool names.
@ -42,13 +71,13 @@ Each server entry specifies a transport type and optional timeouts. The server n
The most common transport. Fabro spawns a child process on the host and communicates over stdin/stdout:
```toml
[mcp_servers.filesystem]
[run.agent.mcps.filesystem]
type = "stdio"
command = ["npx", "-y", "@modelcontextprotocol/server-filesystem", "/workspace"]
startup_timeout_secs = 15
tool_timeout_secs = 90
startup_timeout = "15s"
tool_timeout = "90s"
[mcp_servers.filesystem.env]
[run.agent.mcps.filesystem.env]
NODE_ENV = "production"
```
@ -56,20 +85,21 @@ NODE_ENV = "production"
|---|---|---|
| `type` | Must be `"stdio"`. | — |
| `command` | Array: the executable followed by its arguments. | — |
| `script` | Shell script alternative to `command`. | — |
| `env` | Additional environment variables for the child process. | `{}` |
| `startup_timeout_secs` | Max seconds to wait for the MCP handshake. | `10` |
| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` |
| `startup_timeout` | Max duration to wait for the MCP handshake. | `"10s"` |
| `tool_timeout` | Max duration for a single tool call. | `"60s"` |
### HTTP
For remote MCP servers accessible over Streamable HTTP:
```toml
[mcp_servers.sentry]
[run.agent.mcps.sentry]
type = "http"
url = "https://mcp.sentry.dev/mcp"
[mcp_servers.sentry.headers]
[run.agent.mcps.sentry.headers]
Authorization = "Bearer sk-xxx"
```
@ -78,30 +108,31 @@ Authorization = "Bearer sk-xxx"
| `type` | Must be `"http"`. | — |
| `url` | The MCP server endpoint URL. | — |
| `headers` | Optional HTTP headers (e.g., for authentication). | `{}` |
| `startup_timeout_secs` | Max seconds to wait for the MCP handshake. | `10` |
| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` |
| `startup_timeout` | Max duration to wait for the MCP handshake. | `"10s"` |
| `tool_timeout` | Max duration for a single tool call. | `"60s"` |
### Sandbox
Runs the MCP server inside the workflow's sandbox (e.g., a [Daytona](/integrations/daytona) cloud VM). Fabro starts the server as a background process, waits for it to listen on the specified port, obtains an authenticated preview URL, and connects via HTTP. This is the right transport for MCP servers that need access to the sandbox environment — for example, [Playwright](https://github.com/microsoft/playwright-mcp) for browser automation.
```toml
[mcp_servers.playwright]
[run.agent.mcps.playwright]
type = "sandbox"
command = ["npx", "@playwright/mcp@latest", "--port", "3100", "--headless", "--browser", "chromium"]
port = 3100
startup_timeout_secs = 60
tool_timeout_secs = 120
startup_timeout = "60s"
tool_timeout = "2m"
```
| Field | Description | Default |
|---|---|---|
| `type` | Must be `"sandbox"`. | — |
| `command` | Array: the command to run inside the sandbox. Must include a flag that makes the server listen on `port`. | — |
| `script` | Shell script alternative to `command`. | — |
| `port` | The port the MCP server listens on inside the sandbox. | — |
| `env` | Additional environment variables for the server process. | `{}` |
| `startup_timeout_secs` | Max seconds to wait for the server to start listening and complete the MCP handshake. | `10` |
| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` |
| `startup_timeout` | Max duration to wait for the server to start listening and complete the MCP handshake. | `"10s"` |
| `tool_timeout` | Max duration for a single tool call. | `"60s"` |
The sandbox transport requires a remote sandbox provider (Daytona) that supports preview URLs. During session initialization, Fabro:
@ -117,7 +148,7 @@ The sandbox transport requires a remote sandbox provider (Daytona) that supports
Each MCP server is started sequentially during session initialization. The startup sequence for each server is:
1. **Spawn / connect** — For stdio, spawn the child process. For HTTP, create the HTTP client. For sandbox, start the server inside the sandbox and connect via preview URL.
2. **Handshake** — Perform the MCP protocol handshake within the `startup_timeout_secs` window.
2. **Handshake** — Perform the MCP protocol handshake within the `startup_timeout` window.
3. **Tool discovery** — Call `tools/list` to enumerate available tools.
4. **Registration** — Add each tool to the agent's registry with its qualified name.
@ -132,7 +163,7 @@ When the LLM calls an MCP tool:
3. The server executes the tool and returns a result
4. The result is converted to text and returned to the LLM as a tool result
Tool calls are subject to the `tool_timeout_secs` configured on the server. If a call exceeds the timeout, it fails with a timeout error.
Tool calls are subject to the `tool_timeout` configured on the server. If a call exceeds the timeout, it fails with a timeout error.
### Content handling
@ -197,7 +228,7 @@ MCP servers that fail to start do not block the agent session. The agent proceed
## Protocol details
Fabro implements the MCP client side using the `rmcp` SDK (protocol version `2025-03-26`). The client identifies itself as `fabro-mcp` and supports:
For agent-side MCP connections, Fabro implements the MCP client side using the `rmcp` SDK (protocol version `2025-03-26`). The client identifies itself as `fabro-mcp` and supports:
- Tool listing and invocation
- Server logging notifications (routed to Fabro's tracing system)

View file

@ -127,7 +127,7 @@ The tracked paths are stored as `files_touched` on the stage outcome:
For the **API backend**, Fabro subscribes to agent session events. When a `ToolCallStarted` event fires for `write_file` or `edit_file`, Fabro records the `file_path` argument as pending. When the corresponding `ToolCallCompleted` arrives without an error, the path is confirmed as touched. Failed tool calls are discarded.
For the **CLI backend**, Fabro takes a different approach: it runs `git diff --name-only` and `git ls-files --others --exclude-standard` before and after the agent session, then computes the difference. Any files that appear in the "after" snapshot but not "before" are recorded as touched.
For the **CLI** and **ACP** backends, Fabro takes a different approach: it runs `git diff --name-only` and `git ls-files --others --exclude-standard` before and after the external agent session, then computes the difference. Any files that appear in the "after" snapshot but not "before" are recorded as touched.
## Artifact offloading

View file

@ -6,7 +6,7 @@ description: "Delegate subtasks to child agent sessions"
An agent can spawn **sub-agents** to delegate work to independent child sessions. Each sub-agent gets its own LLM session and tool access, runs concurrently with the parent, and returns its result when finished.
<Note>
Sub-agents are only available with the [API backend](/core-concepts/agents#api-backend-default) (the default). Agents using the [CLI backend](/core-concepts/agents#cli-backend) cannot spawn sub-agents.
Sub-agents are only available with the [API backend](/core-concepts/agents#api-backend-default) (the default). Agents using the [CLI backend](/core-concepts/agents#cli-backend) or [ACP backend](/core-concepts/agents#acp-backend) cannot spawn Fabro sub-agents.
</Note>
## Tools

View file

@ -6,7 +6,7 @@ description: "Built-in tools for file I/O, shell commands, search, and web acces
Every agent in Fabro has access to a set of built-in tools for interacting with the codebase and environment. Tools execute inside the agent's [sandbox](/execution/environments) — whether that's the local machine, a Docker container, or a Daytona VM — so the same tool calls work identically regardless of provider.
<Note>
The tools described on this page apply to the **API backend** (the default). When using the [CLI backend](/core-concepts/agents#cli-backend), the external CLI tool (`claude`, `codex`, or `gemini`) provides its own tools — Fabro's built-in tools are not used.
The tools described on this page apply to the **API backend** (the default). When using the [CLI backend](/core-concepts/agents#cli-backend) or [ACP backend](/core-concepts/agents#acp-backend), the external agent process provides its own tools — Fabro's built-in tools are not used.
</Note>
## Core tools

View file

@ -13,11 +13,11 @@ Two new built-in tools bring real-time information into workflow decisions. `web
## CLI backends
Individual workflow nodes can now delegate work to external AI coding assistants. Set the backend to `claude-code`, `codex`, or `gemini-cli` and the node will use that CLI tool instead of the built-in agent loop.
Individual workflow nodes can delegate work to external AI coding assistants. Set `backend="cli"` and choose a provider; Fabro selects `claude`, `codex`, or `gemini` for the node instead of the built-in API agent loop. Current backend values are `api`, `cli`, and `acp`.
```dot
implement [handler=codergen, cli_backend=codex]
review [handler=codergen, cli_backend=claude-code]
implement [type="agent", backend="cli", provider="openai"]
review [type="agent", backend="cli", provider="anthropic"]
```
This means each stage in a workflow can use a different AI tool — use Codex for implementation and Claude Code for review, for example.

View file

@ -1,5 +1,5 @@
---
title: "Run Events, artifacts, and typed interviews"
title: "Run Events, sandbox lifecycle, and typed interviews"
date: "2026-05-08"
---
@ -18,21 +18,26 @@ Run detail sidebars now include dedicated `Run Events` and `Artifacts` pages. `R
This gives you a direct path to the two things users usually need after a run: the complete audit trail and the files produced by stages.
## Run-owned sandbox lifecycle
Run sandboxes now belong to the run lifecycle instead of being detached provider resources that require manual cleanup. Terminal runs stop their sandboxes by default, resumes reconnect to persisted sandboxes, and run deletion either removes the sandbox or returns preserved provider details according to the run's preserve settings.
The run settings page now shows `stop_on_terminal` alongside sandbox preservation, so you can tell whether cleanup is automatic before launching or deleting a run.
## Typed interview answers
Human-in-the-loop answers now have explicit API shapes for yes/no, single-select, multi-select, and text responses. The web UI and CLI attach flow use those shapes consistently, which removes ambiguity around which field should be present for each question type.
`fabro attach` also handles interview edge cases better. Invalid input re-prompts instead of dropping the interaction, and attach no longer blocks when a question was already answered from another client.
## API client coverage for the web app
The web app now uses the generated TypeScript API client for the auth, workflow, run, artifact, settings, and human-in-the-loop calls that can be generated from OpenAPI. The API reference now documents the browser auth and workflow routes the app relies on, which keeps the UI and public contract aligned as routes evolve.
## More
<Accordion title="API">
- `DELETE /api/v1/runs/{id}` can now return preserved sandbox details when deletion leaves a provider resource alive
- `RunSandboxSettings` now includes `stop_on_terminal`
- `RunStage` now includes a canonical `handler` field so clients can choose agent, command, or debug renderers without guessing from names
- Run summaries now include stored pull request records with `html_url`
- The web app now uses the generated TypeScript API client for auth, workflow, run, artifact, settings, and human-in-the-loop calls generated from OpenAPI
- OpenAPI now documents browser auth config, current user, demo toggle, and development-token login endpoints
- OpenAPI now documents workflow list, workflow detail, and workflow runs endpoints
- `GET /api/v1/runs/{id}/graph` now documents the optional `direction` query parameter
@ -49,6 +54,8 @@ The web app now uses the generated TypeScript API client for the auth, workflow,
<Accordion title="Improvements">
- Added a Profile page to the user menu
- Run titles now render inline Markdown in run lists and run detail headers
- Run Files sidebar now shows the whole tree instead of filtering to changed files only
- Run cards now show a pull request icon and number only when a stored pull request exists
- Run cards now show repository names without repeating the owner prefix
- Archived runs now have a Delete action in the web app
@ -56,6 +63,8 @@ The web app now uses the generated TypeScript API client for the auth, workflow,
</Accordion>
<Accordion title="Fixes">
- Fixed provider token usage normalization so OpenAI, Gemini, and Anthropic totals match provider billing semantics
- Fixed terminal sandbox regressions after run-owned sandbox lifecycle changes
- Fixed prompt-only stages missing completed responses in the Thread view
- Fixed stage interrupt events missing from the Thread view
- Fixed OpenAI reasoning metadata being lost across stateless round trips

View file

@ -0,0 +1,78 @@
---
title: "Sparse inputs, sandbox terminals, and run file diffs"
date: "2026-05-09"
---
<Warning>
**Automatic retros have been removed.** Workflow runs now go directly from execution to finalization and optional pull request creation. The retro crate, retro events, retro docs page, PR retro section, `--no-retro`, `[run.execution].retros`, `features.retros`, and retro projection fields are no longer part of the product surface.
To migrate:
1. Use Run Events, Stages, Turns, logs, and dumps for post-run analysis.
2. Remove retro-specific CLI flags, config keys, API field reads, and docs links.
</Warning>
<Warning>
**Public local worktree mode has been removed.** Local sandbox runs now execute directly in the resolved working directory, and `--in-place` plus `run.sandbox.local.worktree_mode` are no longer supported.
To migrate:
1. Run with `--sandbox local` from the checkout you want Fabro to use.
2. Create a separate clone or Git worktree yourself when you want local isolation.
3. Use Docker or Daytona for managed sandbox isolation.
</Warning>
## Sparse input overrides
Workflow inputs can now be overridden one key at a time from the CLI. Repeat `-I` or `--input` on `fabro run`, `fabro create`, and `fabro preflight` to replace specific inputs while preserving unrelated inherited values from settings and workflow config.
```bash
fabro run .fabro/workflows/check/workflow.toml -I repo_name=fabro-2 --input language=rust
```
Input-driven prompt paths, imports, and child workflow paths are rendered with the effective inputs before bundling, so sparse overrides work even when they change which files a workflow references.
## Sandbox access from run detail
Run detail now has a dedicated Sandbox tab and terminal route. You can inspect the run sandbox from the web app, open an interactive terminal, copy Docker exec access commands, and see sandbox identity information without switching to a separate CLI session.
This also gives long-running debug sessions a clearer place to live. Terminal framing, scroll behavior, Daytona proxying, and terminal error states were tightened so the terminal stays usable inside the run UI.
## Run file diff controls
The Files Changed view can now compare committed run history and sandbox file scopes from the same toolbar. You can switch diff scope next to the file count, pick commits from the run history, and refresh patch diffs when the selected scope changes.
This makes the file browser useful both during active sandbox work and after a run has committed checkpoints.
## More
<Accordion title="API">
- Run file APIs now expose commit lists and diff scope metadata for committed history and sandbox comparisons
- Run payloads, events, and mutations now support explicit run titles
- Billing projections now include live per-stage token usage while a run is active
- Sandbox details now expose control-plane metadata used by the run Sandbox tab
</Accordion>
<Accordion title="CLI">
- Added repeatable `-I, --input <key=value>` to `fabro run`, `fabro create`, and `fabro preflight`
- Added Docker host preflight checks for Docker-based deployment and sandbox diagnostics
</Accordion>
<Accordion title="Improvements">
- Run titles can now be edited inline from the run header
- Run title fields now preserve explicit titles across create, fork, archive, unarchive, and attach flows
- Added a demo-only Start tab for app-shell demos
- Docker Compose deployment defaults now include safer sandbox-facing defaults
- Added Thread DNA timeline strips to debug events and agent stage views
- Added specialized stage renderers for non-agent workflow handlers
- The Billing tab now shows an empty state when no models were used
</Accordion>
<Accordion title="Fixes">
- Fixed deletion of unreadable runs
- Fixed CLI verification failure handling
- Fixed Daytona terminals to use the toolbox proxy and hide control frames
- Fixed terminal dock spacing, bottom-row clipping, and overlap with the steer bar
- Fixed run file diffs refreshing when the selected scope changes
- Fixed scoped run file diffs to only include tracked files
- Fixed deprecated project directory settings being applied to workflow discovery
- Fixed embedded build Git metadata being stale on branch commits
</Accordion>

View file

@ -0,0 +1,57 @@
---
title: "Sandbox tools, auth sessions, and Live Events"
date: "2026-05-10"
---
<Warning>
**Run API responses now use the canonical `Run` payload.** Run list, board, create, retrieve, and lifecycle endpoints now return the same public run shape. Archive state moved out of `status.kind = "archived"` and into run lifecycle metadata, sandbox fields distinguish planned settings from runtime state, and pull request records are separate from live pull request details.
To migrate:
1. Regenerate API clients from the current OpenAPI spec.
2. Replace `RunSummary`, `RunListItem`, and `RunStatusResponse` assumptions with the canonical `Run` shape.
3. Read archive state from lifecycle metadata instead of treating `archived` as a terminal status.
</Warning>
## Sandbox tools in one place
The run Sandbox tab now includes a read-only filesystem browser, Daytona VNC previews, discovered sandbox services, and terminal access from the same page. Empty files render cleanly, large file previews are virtualized, and directory sentinel files are hidden from the browser.
Daytona sandboxes can start a signed noVNC preview from the web UI, and discovered listening TCP services are listed with their ports, bind addresses, process summaries, and preview support.
## Auth sessions and Live Events
Profile now includes a Sessions page for the current browser session and active CLI session chains. CLI sessions can be revoked from the same surface, while browser sessions are listed as non-revocable in this API version.
Settings now includes a Live Events page for watching server events in the web app. The page reuses the event debugger and adds the filtering and navigation needed for operational inspection.
## Automations navigation
The web app now uses Automations as the product label for workflow definitions and runs. Routes, nav labels, and run pages were updated together so the main app shell reads as Automations, Runs, Settings, and Profile instead of mixing workflow terminology into the navigation.
## More
<Accordion title="API">
- New `GET /api/v1/auth/sessions` endpoint lists browser and CLI auth sessions for the authenticated user
- New `DELETE /api/v1/auth/sessions/{id}` endpoint revokes active CLI session chains
- New `GET /api/v1/runs/{id}/sandbox/services` endpoint lists listening TCP services inside a run sandbox
- New `POST /api/v1/runs/{id}/sandbox/vnc` endpoint creates signed noVNC preview URLs for Daytona sandboxes
- Run list, board, create, retrieve, and lifecycle endpoints now return canonical `Run` objects
</Accordion>
<Accordion title="Improvements">
- Added Profile sub-navigation with Overview and Sessions pages
- Added Settings sub-navigation with Overview and Live Events pages
- Added a full-screen terminal route and an "Open in new tab" action for embedded terminals
- Added read-only filesystem and Daytona VNC modes to the run Sandbox tab
- Added a Services tab to the run Sandbox page
- Virtualized sandbox file previews and improved empty-file handling
- Renamed the Workflows tab to Automations
</Accordion>
<Accordion title="Fixes">
- Fixed sandbox service discovery so listening services are detected more reliably
- Fixed sandbox VNC previews to open the noVNC viewer page
- Fixed full-screen terminal toast notifications by wrapping the route in the Toast provider
- Fixed filesystem directory sentinels showing in the sandbox file browser
- Fixed projection cache hydration before appending later events
</Accordion>

View file

@ -0,0 +1,47 @@
---
title: "Fabro MCP server"
date: "2026-05-11"
---
## Fabro MCP server
Fabro now ships a stdio-based Model Context Protocol server, so MCP clients can manage workflow runs through the authenticated `fabro` CLI. It reuses normal CLI server targeting, OAuth refresh, dev-token and local-server handling, proxy behavior, and storage configuration instead of requiring a separate MCP authentication flow.
```bash
fabro mcp init claude
```
You can also print client configuration JSON or start the server directly:
```bash
fabro mcp config
fabro mcp start
```
## Run management from MCP clients
The MCP server exposes structured tools for creating runs, searching runs, reading run events, interacting with pending human questions, and waiting for runs to finish. MCP-created runs use the same manifest construction and override semantics as CLI-created runs, so workflow paths, goals, inputs, labels, model settings, sandbox settings, and dry-run options behave consistently.
This lets agent tools orchestrate Fabro runs without scraping CLI output or hand-rolling HTTP clients.
## More
<Accordion title="CLI">
- Added `fabro mcp start` to launch the MCP server over stdio
- Added `fabro mcp config` to print MCP client configuration JSON
- Added `fabro mcp init <agent>` for Claude, Cursor, and Windsurf client setup
</Accordion>
<Accordion title="Workflows">
- Added a bundled `daytona-medium` workflow for verifying the Daytona Medium sandbox starts with standard tooling
</Accordion>
<Accordion title="Improvements">
- MCP run tools can create multiple runs, apply scalar input overrides, attach labels, start or stage runs, and return structured run summaries
- MCP event tools support category, event type, text, timestamp, direction, and pagination filters
- MCP interact tools support cancelling, archiving, unarchiving, inspecting questions, answering questions, and sending run messages
</Accordion>
<Accordion title="Fixes">
- Fixed deleting terminal runs so successful completions are not followed by cancelled failure events
</Accordion>

View file

@ -18,7 +18,7 @@ This loop continues until the model stops calling tools, indicating it considers
## Backends
Every agent node uses a **backend** that determines how Fabro interacts with the LLM. There are two options:
Every agent and prompt node uses a **backend** that determines how Fabro interacts with the LLM. There are three options:
### API backend (default)
@ -31,7 +31,7 @@ Fabro manages the agent loop directly — it calls the LLM provider's API, execu
### CLI backend
Fabro delegates execution to an external coding assistant CLI. The CLI tool manages its own tool loop internally — Fabro sends the prompt, waits for the CLI to finish, and tracks file changes via `git diff` before and after execution.
Fabro delegates execution to a legacy external coding assistant CLI. The CLI tool manages its own tool loop internally — Fabro sends the prompt, waits for the CLI to finish, and tracks file changes via `git diff` before and after execution.
The CLI is selected automatically based on the node's provider:
@ -41,6 +41,8 @@ The CLI is selected automatically based on the node's provider:
| OpenAI | `codex` |
| Gemini | `gemini` |
Fabro does not install these CLIs at runtime. Install the selected CLI in the sandbox image or run setup steps before the workflow reaches a `backend="cli"` node.
Set the CLI backend on a node with `backend="cli"` or via a [model stylesheet](/workflows/stylesheets):
```dot
@ -52,15 +54,29 @@ implement [label="Implement", backend="cli"]
* { backend: cli; }
```
### ACP backend
Fabro can also run Agent Client Protocol (ACP) stdio agents with `backend="acp"`. ACP agents run inside the active Fabro sandbox, so local and Docker runs keep the same workspace isolation, secret forwarding, cancellation, and file-change tracking behavior as other agent stages.
Set ACP on a node with `backend="acp"` and an explicit `acp_command`:
```dot
implement [label="Implement", backend="acp", acp_command="python3 tools/fake_acp_agent.py"]
```
Fabro does not install ACP agents, Node.js, npm, or `npx` at runtime. The command must already be available in the sandbox image, repository, or setup steps. You can use `npx ...@latest` as an explicit `acp_command` if that is the behavior you want, but Fabro will treat it like any other user-supplied command.
ACP v1 does not have a portable model-selection request. Fabro records the selected provider and model in events and run projections, but model-specific ACP behavior must be encoded in the chosen command for now. ACP is supported with local and Docker sandboxes; Daytona does not expose bidirectional stdio yet, so ACP nodes fail there with an explicit unsupported-provider error.
### Comparison
| Capability | API backend | CLI backend |
|---|---|---|
| Tools | Fabro built-in tools + MCP | CLI's own tool set |
| Session caching | Supported (`fidelity` + `thread_id`) | Not supported |
| Sub-agents | Supported | Not supported |
| Provider failover | Supported | Not supported |
| File tracking | Tool call events | `git diff` before/after |
| Capability | API backend | CLI backend | ACP backend |
|---|---|---|---|
| Tools | Fabro built-in tools + MCP | CLI's own tool set | ACP agent's own tool set |
| Session caching | Supported (`fidelity` + `thread_id`) | Not supported | Agent-dependent |
| Sub-agents | Supported | Not supported | Not supported through Fabro tools |
| Provider failover | Supported | Not supported | Not supported |
| File tracking | Tool call events | `git diff` before/after | `git diff` before/after |
### When to use the CLI backend
@ -68,6 +84,12 @@ implement [label="Implement", backend="cli"]
- **CLI-only models** — use models that are only available through a CLI tool, not via API
- **Existing workflows** — integrate a CLI tool you already depend on without rewriting its configuration
### When to use the ACP backend
- **Protocol adapters** — run ACP-compatible coding agents through a stable stdio protocol
- **Sandbox parity** — keep agent process execution inside Fabro's local or Docker sandbox
- **Custom agents** — use `acp_command` for a checked-in or preinstalled ACP adapter
## Tools
Agents have access to a set of built-in tools for interacting with the codebase and environment:

View file

@ -250,6 +250,9 @@
"group": "May 2026",
"icon": "clock-rotate-left",
"pages": [
"changelog/2026-05-11",
"changelog/2026-05-10",
"changelog/2026-05-09",
"changelog/2026-05-08",
"changelog/2026-05-07",
"changelog/2026-05-06",

View file

@ -79,6 +79,7 @@ fabro [OPTIONS] [COMMAND]
| `fabro inspect` | Show detailed information about a workflow run |
| `fabro install` | Set up the Fabro environment (LLMs, certs, GitHub) |
| `fabro logs` | View the raw worker tracing log of a workflow run |
| `fabro mcp` | Model Context Protocol server |
| `fabro model` | List and test LLM models |
| `fabro pr` | Pull request operations |
| `fabro preflight` | Validate run configuration without executing |
@ -510,6 +511,73 @@ fabro logs [OPTIONS] <RUN>
| `--server <server>` | Fabro server target: http(s) URL or absolute Unix socket path |
| `-n, --tail <tail>` | Lines from end (default: all) |
### `fabro mcp`
Model Context Protocol server
```bash
fabro mcp [OPTIONS] <COMMAND>
```
#### Subcommands
| Command | Description |
| --- | --- |
| `fabro mcp config` | Print MCP client configuration JSON |
| `fabro mcp init` | Configure an MCP client to launch Fabro |
| `fabro mcp start` | Start the Fabro MCP server over stdio |
#### `fabro mcp config`
Print MCP client configuration JSON
```bash
fabro mcp config [OPTIONS]
```
#### Options
| Option | Description |
| --- | --- |
| `--server <server>` | Fabro server target: http(s) URL or absolute Unix socket path |
| `--storage-dir <storage_dir>` | Local storage directory (default: ~/.fabro/storage) |
#### `fabro mcp init`
Configure an MCP client to launch Fabro
```bash
fabro mcp init [OPTIONS] <AGENT>
```
#### Arguments
| Name | Description |
| --- | --- |
| `AGENT` | Values: `claude`, `cursor`, `windsurf` |
#### Options
| Option | Description |
| --- | --- |
| `--server <server>` | Fabro server target: http(s) URL or absolute Unix socket path |
| `--storage-dir <storage_dir>` | Local storage directory (default: ~/.fabro/storage) |
#### `fabro mcp start`
Start the Fabro MCP server over stdio
```bash
fabro mcp start [OPTIONS]
```
#### Options
| Option | Description |
| --- | --- |
| `--server <server>` | Fabro server target: http(s) URL or absolute Unix socket path |
| `--storage-dir <storage_dir>` | Local storage directory (default: ~/.fabro/storage) |
### `fabro model`
List and test LLM models

View file

@ -206,7 +206,8 @@ Start nodes can also be identified by ID (`start` or `Start`). Exit nodes can be
| `model` | String | Explicit model ID (overrides stylesheet) |
| `provider` | String | Explicit provider name (overrides stylesheet). Auto-inferred from the model catalog when omitted. |
| `project_memory` | Boolean | When `true` (default), prompt nodes discover and include project docs (`AGENTS.md`, `CLAUDE.md`, etc.) as a system prompt. Set to `false` to disable. |
| `backend` | String | Agent execution backend. `api` (default): Fabro calls the LLM API directly and runs its own tool loop. `cli`: Fabro delegates to an external CLI tool (`claude`, `codex`, or `gemini` based on provider). See [Agents — Backends](/core-concepts/agents#backends). |
| `backend` | String | Agent execution backend: `api` (default), `cli`, or `acp`. `api` runs Fabro's tool loop through provider APIs; `cli` delegates to the legacy provider CLI; `acp` runs an Agent Client Protocol stdio agent inside the active sandbox. See [Agents — Backends](/core-concepts/agents#backends). |
| `acp_command` | String | Required for nodes with `backend="acp"`. The value must be a stdio ACP command available in the sandbox. Fabro records model selection but does not send it through stable ACP v1. |
### Command nodes

View file

@ -88,7 +88,7 @@ Stylesheets support four properties:
| `model` | Model ID or alias (e.g. `claude-sonnet-4-5`, `opus`, `gemini-pro`) |
| `provider` | Provider name (optional — auto-inferred from the model catalog when omitted) |
| `reasoning_effort` | `low`, `medium`, or `high` |
| `backend` | `api` (default) or `cli` |
| `backend` | `api` (default), `cli`, or `acp` |
## Why route models?

View file

@ -68,7 +68,7 @@ Stylesheets support four properties:
| `provider` | Provider name (optional — auto-inferred from the model catalog when omitted) | `anthropic`, `openai`, `gemini` |
| `reasoning_effort` | Reasoning effort level | `low`, `medium`, `high` |
| `speed` | Output speed mode. `fast` enables Anthropic's fast mode for up to 2.5x faster output at higher cost. | `fast` |
| `backend` | Agent execution backend — `api` (default) runs Fabro's own tool loop, `cli` delegates to an external CLI tool. See [Backends](/core-concepts/agents#backends). | `cli`, `api` |
| `backend` | Agent execution backend — `api` (default) runs Fabro's own tool loop, `cli` delegates to a legacy external CLI tool, and `acp` runs an Agent Client Protocol stdio agent in the active sandbox. See [Backends](/core-concepts/agents#backends). | `api`, `cli`, `acp` |
See [Models](/core-concepts/models) for the full list of model IDs and aliases.

View file

@ -0,0 +1,37 @@
[package]
name = "fabro-acp"
edition.workspace = true
version.workspace = true
publish = false
license.workspace = true
description = "Agent Client Protocol backend support for Fabro"
[features]
test-support = []
[lib]
doctest = false
[lints]
workspace = true
[dependencies]
agent-client-protocol.workspace = true
agent-client-protocol-tokio.workspace = true
fabro-model = { path = "../fabro-model" }
fabro-sandbox = { path = "../fabro-sandbox" }
fabro-types = { path = "../fabro-types" }
fabro-util = { path = "../fabro-util" }
bytes.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
tokio.workspace = true
tokio-util = { workspace = true, features = ["compat", "io"] }
futures.workspace = true
uuid.workspace = true
tracing.workspace = true
[dev-dependencies]
fabro-sandbox = { path = "../fabro-sandbox", features = ["test-support"] }
tempfile = "3"

View file

@ -0,0 +1,190 @@
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::str::FromStr;
use agent_client_protocol::schema::McpServer;
use agent_client_protocol_tokio::AcpAgent;
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct AcpCommand {
display: String,
program: PathBuf,
args: Vec<String>,
env: HashMap<String, String>,
}
impl AcpCommand {
#[must_use]
pub fn program(&self) -> &Path {
&self.program
}
#[must_use]
pub fn args(&self) -> &[String] {
&self.args
}
#[must_use]
pub fn env(&self) -> &HashMap<String, String> {
&self.env
}
#[must_use]
pub fn display(&self) -> &str {
&self.display
}
#[must_use]
pub fn to_shell_command(&self) -> String {
render_command(&self.program, &self.args)
}
}
impl std::fmt::Display for AcpCommand {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(&self.display)
}
}
#[derive(Debug, thiserror::Error)]
pub enum AcpCommandError {
#[error("acp_command must not be empty")]
EmptyOverride,
#[error(
"acp_command is required for backend=\"acp\" because Fabro does not install ACP agents"
)]
MissingOverride,
#[error("only stdio ACP commands are supported")]
UnsupportedTransport,
#[error("failed to parse acp_command")]
Parse(#[source] agent_client_protocol::Error),
}
impl From<agent_client_protocol::Error> for AcpCommandError {
fn from(error: agent_client_protocol::Error) -> Self {
Self::Parse(error)
}
}
pub fn resolve_acp_command(override_command: Option<&str>) -> Result<AcpCommand, AcpCommandError> {
if let Some(raw) = override_command {
let trimmed = raw.trim();
if trimmed.is_empty() {
return Err(AcpCommandError::EmptyOverride);
}
return parse_acp_command(trimmed);
}
Err(AcpCommandError::MissingOverride)
}
fn parse_acp_command(raw: &str) -> Result<AcpCommand, AcpCommandError> {
reject_non_stdio_json_transport(raw)?;
let agent = AcpAgent::from_str(raw)?;
let McpServer::Stdio(stdio) = agent.into_server() else {
return Err(AcpCommandError::UnsupportedTransport);
};
let program = stdio.command;
let args = stdio.args;
let display = render_command(&program, &args);
Ok(AcpCommand {
display,
program,
args,
env: stdio
.env
.into_iter()
.map(|env| (env.name, env.value))
.collect(),
})
}
fn render_command(program: &Path, args: &[String]) -> String {
std::iter::once(program.to_string_lossy().into_owned())
.chain(args.iter().cloned())
.map(|part| fabro_sandbox::shell_quote(&part))
.collect::<Vec<_>>()
.join(" ")
}
fn reject_non_stdio_json_transport(raw: &str) -> Result<(), AcpCommandError> {
let trimmed = raw.trim_start();
if !trimmed.starts_with('{') {
return Ok(());
}
let Ok(value) = serde_json::from_str::<serde_json::Value>(trimmed) else {
return Ok(());
};
match value.get("type").and_then(serde_json::Value::as_str) {
Some("stdio") | None => Ok(()),
Some(_) => Err(AcpCommandError::UnsupportedTransport),
}
}
#[cfg(test)]
mod tests {
use std::path::Path;
use super::*;
#[test]
fn missing_acp_command_is_rejected() {
let err = resolve_acp_command(None).unwrap_err();
assert!(
err.to_string()
.contains("acp_command is required for backend=\"acp\"")
);
}
#[test]
fn explicit_acp_command_overrides_provider_default() {
let command = resolve_acp_command(Some("python fake_agent.py")).unwrap();
assert_eq!(command.to_string(), "python fake_agent.py");
assert_eq!(command.program(), Path::new("python"));
assert_eq!(command.args(), &["fake_agent.py".to_string()]);
}
#[test]
fn blank_acp_command_is_rejected() {
let err = resolve_acp_command(Some(" ")).unwrap_err();
assert!(err.to_string().contains("acp_command must not be empty"));
}
#[test]
fn json_stdio_acp_command_is_supported() {
let raw = r#"{"type":"stdio","name":"fake","command":"python","args":["fake agent.py"],"env":[{"name":"MODE","value":"test"}]}"#;
let command = resolve_acp_command(Some(raw)).unwrap();
assert_eq!(command.program(), Path::new("python"));
assert_eq!(command.args(), &["fake agent.py".to_string()]);
assert_eq!(command.env().get("MODE").map(String::as_str), Some("test"));
}
#[test]
fn json_stdio_acp_command_display_omits_env_contents() {
let raw = r#"{"type":"stdio","name":"fake","command":"agent","args":["--flag","two words"],"env":[{"name":"OPENAI_API_KEY","value":"secret-key"}]}"#;
let command = resolve_acp_command(Some(raw)).unwrap();
assert_eq!(
command.env().get("OPENAI_API_KEY").map(String::as_str),
Some("secret-key")
);
assert_eq!(command.to_string(), "agent --flag 'two words'");
assert!(!command.to_string().contains("secret-key"));
assert!(!command.to_string().contains("OPENAI_API_KEY"));
}
#[test]
fn non_stdio_acp_command_is_rejected() {
let raw = r#"{"type":"http","name":"remote","url":"https://example.test/acp"}"#;
let err = resolve_acp_command(Some(raw)).unwrap_err();
assert!(
err.to_string()
.contains("only stdio ACP commands are supported")
);
}
}

View file

@ -0,0 +1,31 @@
use crate::command::AcpCommandError;
#[derive(Debug, thiserror::Error)]
pub enum AcpError {
#[error(transparent)]
Command(#[from] AcpCommandError),
#[error(transparent)]
Sandbox(#[from] fabro_sandbox::Error),
#[error("ACP protocol error")]
Protocol(#[source] agent_client_protocol::Error),
#[error("ACP turn was cancelled")]
Cancelled,
#[error("ACP turn timed out")]
TimedOut { stderr: String },
#[error("ACP prompt stopped with {stop_reason}: {text}")]
StopReason {
stop_reason: String,
text: String,
},
}
impl From<agent_client_protocol::Error> for AcpError {
fn from(error: agent_client_protocol::Error) -> Self {
Self::Protocol(error)
}
}

View file

@ -0,0 +1,12 @@
pub mod command;
pub mod error;
pub mod session;
#[cfg(any(test, feature = "test-support"))]
pub mod test_support;
mod transport;
pub use command::{AcpCommand, AcpCommandError, resolve_acp_command};
pub use error::AcpError;
pub use session::{AcpRunRequest, AcpRunResult, render_stop_reason, run_acp_turn};

View file

@ -0,0 +1,268 @@
use std::collections::HashMap;
use std::sync::Arc;
use std::time::Duration;
use agent_client_protocol::schema::{
CancelNotification, ContentBlock, ContentChunk, InitializeRequest, PermissionOptionKind,
ProtocolVersion, RequestPermissionOutcome, RequestPermissionRequest, RequestPermissionResponse,
SelectedPermissionOutcome, SessionNotification, SessionUpdate, StopReason,
};
use agent_client_protocol::util::MatchDispatch;
use agent_client_protocol::{ActiveSession, Agent, Client, Error as ProtocolError, SessionMessage};
use fabro_sandbox::Sandbox;
use fabro_util::time::elapsed_ms;
use tokio::time::{sleep, timeout};
use tokio_util::sync::CancellationToken;
use crate::command::AcpCommand;
use crate::error::AcpError;
use crate::transport::{SandboxAcpTransport, TransportState};
pub struct AcpRunRequest {
pub command: AcpCommand,
pub prompt: String,
pub cwd: String,
pub timeout_ms: Option<u64>,
pub env: HashMap<String, String>,
pub sandbox: Arc<dyn Sandbox>,
pub cancel_token: CancellationToken,
pub on_activity: Option<Arc<dyn Fn() + Send + Sync>>,
}
#[derive(Debug)]
pub struct AcpRunResult {
pub text: String,
pub stop_reason: StopReason,
pub stderr: String,
pub duration_ms: u64,
}
pub async fn run_acp_turn(request: AcpRunRequest) -> Result<AcpRunResult, AcpError> {
let AcpRunRequest {
command,
prompt,
cwd,
timeout_ms,
env,
sandbox,
cancel_token,
on_activity,
} = request;
let start = std::time::Instant::now();
let state = TransportState::new();
let read_cancel_token = cancel_token.clone();
let run_cancel_token = cancel_token.clone();
let permission_cancel_token = cancel_token.clone();
let state_for_run = state.clone();
let transport = SandboxAcpTransport::new(command, cwd.clone(), env, sandbox, state.clone());
let run = Client
.builder()
.name("fabro")
.on_receive_request(
async move |request: RequestPermissionRequest, responder, _connection| {
let outcome = if permission_cancel_token.is_cancelled() {
RequestPermissionOutcome::Cancelled
} else {
select_permission_outcome(&request)
};
responder.respond(RequestPermissionResponse::new(outcome))
},
agent_client_protocol::on_receive_request!(),
)
.connect_with(transport, async move |cx| {
cx.send_request(InitializeRequest::new(ProtocolVersion::V1))
.block_task()
.await?;
cx.build_session(&cwd)
.block_task()
.run_until(async |mut session| {
session.send_prompt(prompt)?;
read_turn(
&mut session,
&read_cancel_token,
on_activity.as_ref(),
&state_for_run,
)
.await
})
.await
});
let cancel_deadline_token = cancel_token.clone();
let run_outcome = async {
match timeout_ms {
Some(timeout_ms) => {
if let Ok(result) = timeout(Duration::from_millis(timeout_ms), run).await {
Ok(result)
} else {
state.terminate().await?;
if run_cancel_token.is_cancelled() {
return Err(AcpError::Cancelled);
}
Err(AcpError::TimedOut {
stderr: state.stderr_tail().await,
})
}
}
None => Ok(run.await),
}
};
let outcome = tokio::select! {
result = run_outcome => result?,
() = async {
cancel_deadline_token.cancelled().await;
sleep(Duration::from_millis(500)).await;
} => {
state.terminate().await?;
return Err(AcpError::Cancelled);
}
};
let (text, stop_reason) = match outcome {
Ok(result) => result,
Err(_) if run_cancel_token.is_cancelled() => {
state.terminate().await?;
return Err(AcpError::Cancelled);
}
Err(error) => {
state.terminate().await?;
if let Some(startup_error) = state.take_startup_error().await {
return Err(AcpError::Sandbox(startup_error));
}
return Err(map_protocol_error(error));
}
};
match stop_reason {
StopReason::EndTurn | StopReason::Refusal => {}
StopReason::Cancelled => {
state.terminate().await?;
return Err(AcpError::Cancelled);
}
_ => {
state.terminate().await?;
return Err(AcpError::StopReason {
stop_reason: render_stop_reason(&stop_reason),
text,
});
}
}
state.terminate().await?;
let stderr = state.stderr_tail().await;
Ok(AcpRunResult {
text,
stop_reason,
stderr,
duration_ms: elapsed_ms(start),
})
}
fn map_protocol_error(error: ProtocolError) -> AcpError {
AcpError::Protocol(error)
}
fn select_permission_outcome(request: &RequestPermissionRequest) -> RequestPermissionOutcome {
let selected = request
.options
.iter()
.find(|option| option.kind == PermissionOptionKind::AllowAlways)
.or_else(|| {
request
.options
.iter()
.find(|option| option.kind == PermissionOptionKind::AllowOnce)
})
.or_else(|| {
request.options.iter().find(|option| {
!matches!(
option.kind,
PermissionOptionKind::RejectOnce | PermissionOptionKind::RejectAlways
)
})
});
selected.map_or(RequestPermissionOutcome::Cancelled, |option| {
RequestPermissionOutcome::Selected(SelectedPermissionOutcome::new(option.option_id.clone()))
})
}
async fn read_turn(
session: &mut ActiveSession<'_, Agent>,
cancel_token: &CancellationToken,
on_activity: Option<&Arc<dyn Fn() + Send + Sync>>,
state: &TransportState,
) -> Result<(String, StopReason), ProtocolError> {
let mut text = String::new();
let mut cancel_sent = false;
loop {
tokio::select! {
update = session.read_update() => {
if let Some(on_activity) = on_activity {
on_activity();
}
match update? {
SessionMessage::SessionMessage(dispatch) => {
MatchDispatch::new(dispatch)
.if_notification(async |notification: SessionNotification| {
if let SessionUpdate::AgentMessageChunk(ContentChunk {
content: ContentBlock::Text(text_chunk),
..
}) = notification.update {
text.push_str(&text_chunk.text);
}
Ok(())
})
.await
.otherwise_ignore()?;
}
SessionMessage::StopReason(stop_reason) => {
return Ok((text, stop_reason));
}
_ => {}
}
}
() = cancel_token.cancelled(), if !cancel_sent => {
cancel_sent = true;
session.connection().send_notification_to(
Agent,
CancelNotification::new(session.session_id().clone()),
)?;
}
() = sleep(Duration::from_millis(500)), if cancel_sent => {
state.terminate().await.map_err(ProtocolError::into_internal_error)?;
return Ok((text, StopReason::Cancelled));
}
}
}
}
#[must_use]
pub fn render_stop_reason(stop_reason: &StopReason) -> String {
serde_json::to_value(stop_reason)
.ok()
.and_then(|value| value.as_str().map(str::to_string))
.unwrap_or_else(|| format!("{stop_reason:?}"))
}
#[cfg(test)]
mod tests {
use agent_client_protocol::schema::SessionNotification;
#[test]
fn codex_usage_update_session_notification_deserializes() {
let notification = serde_json::json!({
"sessionId": "session-1",
"update": {
"sessionUpdate": "usage_update",
"used": 26128,
"size": 258_400
}
});
serde_json::from_value::<SessionNotification>(notification)
.expect("Codex ACP usage_update notifications should be ignored, not fatal");
}
}

View file

@ -0,0 +1,150 @@
use agent_client_protocol::schema::{
ContentBlock, ContentChunk, SessionNotification, SessionUpdate,
};
use serde_json::json;
pub const SESSION_ID: &str = "sess-1";
pub fn agent_message_chunk(session_id: &str, text: &str) -> SessionNotification {
SessionNotification::new(
session_id.to_string(),
SessionUpdate::AgentMessageChunk(ContentChunk::new(ContentBlock::from(text.to_string()))),
)
}
pub fn agent_message_chunk_json(session_id: &str, text: &str) -> serde_json::Value {
json!({
"jsonrpc": "2.0",
"method": "session/update",
"params": agent_message_chunk(session_id, text),
})
}
pub fn fake_acp_agent_script() -> &'static str {
r#"
import json
import os
import signal
import sys
import time
methods = []
session_id = "sess-1"
if os.environ.get("ACP_PID_RECORD"):
with open(os.environ["ACP_PID_RECORD"], "w", encoding="utf-8") as record:
record.write(str(os.getpid()))
def handle_sigterm(signum, frame):
if os.environ.get("ACP_LINGER_TERMINATED"):
with open(os.environ["ACP_LINGER_TERMINATED"], "w", encoding="utf-8") as record:
record.write("terminated\n")
sys.exit(0)
signal.signal(signal.SIGTERM, handle_sigterm)
def send(message):
print(json.dumps(message), flush=True)
def respond(message, result):
send({"jsonrpc": "2.0", "id": message["id"], "result": result})
def record_methods():
if os.environ.get("ACP_RECORD"):
with open(os.environ["ACP_RECORD"], "w", encoding="utf-8") as record:
record.write("\n".join(methods) + "\n")
for line in sys.stdin:
message = json.loads(line)
method = message.get("method")
methods.append(method)
if method == "initialize":
if os.environ.get("ACP_MODE") == "slow_initialize":
time.sleep(60)
respond(message, {"protocolVersion": 1, "agentCapabilities": {}})
elif method == "session/new":
if os.environ.get("ACP_SESSION_NEW_PARAMS"):
with open(os.environ["ACP_SESSION_NEW_PARAMS"], "w", encoding="utf-8") as record:
record.write(json.dumps(message.get("params", {}), separators=(",", ":")))
respond(message, {"sessionId": session_id})
elif method == "session/prompt":
if os.environ.get("ACP_PROMPT_RECORD"):
with open(os.environ["ACP_PROMPT_RECORD"], "w", encoding="utf-8") as record:
record.write(json.dumps(message.get("params", {})))
mode = os.environ.get("ACP_MODE", "normal")
if mode == "timeout":
time.sleep(60)
if mode == "malformed":
print("malformed json", file=sys.stderr, flush=True)
print("{not-json", flush=True)
break
if mode == "early_exit":
print("early boom", file=sys.stderr, flush=True)
sys.exit(2)
if mode == "write_file":
with open("hello.txt", "w", encoding="utf-8") as file:
file.write("hello from sandbox\n")
if mode == "cancel":
send({
"jsonrpc": "2.0",
"method": "session/update",
"params": {
"sessionId": session_id,
"update": {
"sessionUpdate": "agent_message_chunk",
"content": {"type": "text", "text": "waiting for cancellation"}
}
}
})
for cancel_line in sys.stdin:
cancel_message = json.loads(cancel_line)
if cancel_message.get("method") == "session/cancel":
with open(os.environ["ACP_CANCEL_RECORD"], "w", encoding="utf-8") as record:
record.write("session/cancel\n")
respond(message, {"stopReason": "cancelled"})
sys.exit(0)
if mode == "permission":
send({
"jsonrpc": "2.0",
"id": "permission-1",
"method": "session/request_permission",
"params": {
"sessionId": session_id,
"toolCall": {"toolCallId": "tool-1"},
"options": [
{"optionId": "reject", "name": "Reject", "kind": "reject_once"},
{"optionId": "once", "name": "Allow once", "kind": "allow_once"},
{"optionId": "always", "name": "Allow always", "kind": "allow_always"}
]
}
})
permission_response = json.loads(sys.stdin.readline())
with open(os.environ["ACP_PERMISSION"], "w", encoding="utf-8") as permission:
permission.write(json.dumps(permission_response.get("result", {}), separators=(",", ":")))
for text in ["hello ", "from acp"]:
send({
"jsonrpc": "2.0",
"method": "session/update",
"params": {
"sessionId": session_id,
"update": {
"sessionUpdate": "agent_message_chunk",
"content": {"type": "text", "text": text}
}
}
})
record_methods()
respond(message, {"stopReason": os.environ.get("ACP_STOP_REASON", "end_turn")})
if mode == "linger_after_response":
while True:
time.sleep(1)
break
else:
send({
"jsonrpc": "2.0",
"id": message.get("id"),
"error": {"code": -32601, "message": "method not found"}
})
"#
}

View file

@ -0,0 +1,156 @@
use std::collections::HashMap;
use std::io::Result as IoResult;
use std::pin::Pin;
use std::sync::Arc;
use std::time::Duration;
use agent_client_protocol::util::internal_error;
use agent_client_protocol::{
Agent, Client, ConnectTo, Error as ProtocolError, Lines, Result as AcpProtocolResult,
};
use fabro_sandbox::{
Error as SandboxError, Result as SandboxResult, Sandbox, StderrCollector, StdioProcessHandle,
};
use futures::io::BufReader;
use futures::sink::unfold;
use futures::{AsyncBufReadExt, AsyncWriteExt, Stream};
use tokio::sync::Mutex as TokioMutex;
use tokio::time::timeout;
use tokio_util::compat::{TokioAsyncReadCompatExt, TokioAsyncWriteCompatExt};
use crate::command::AcpCommand;
#[derive(Clone)]
pub(crate) struct TransportState {
handle: Arc<TokioMutex<Option<StdioProcessHandle>>>,
stderr: Arc<TokioMutex<Option<StderrCollector>>>,
startup_error: Arc<TokioMutex<Option<SandboxError>>>,
}
impl TransportState {
pub(crate) fn new() -> Self {
Self {
handle: Arc::new(TokioMutex::new(None)),
stderr: Arc::new(TokioMutex::new(None)),
startup_error: Arc::new(TokioMutex::new(None)),
}
}
async fn set_process(&self, handle: StdioProcessHandle, stderr: StderrCollector) {
*self.handle.lock().await = Some(handle);
*self.stderr.lock().await = Some(stderr);
}
async fn set_startup_error(&self, error: SandboxError) {
*self.startup_error.lock().await = Some(error);
}
pub(crate) async fn take_startup_error(&self) -> Option<SandboxError> {
self.startup_error.lock().await.take()
}
pub(crate) async fn terminate(&self) -> SandboxResult<()> {
if let Some(handle) = self.handle.lock().await.as_ref().cloned() {
handle.terminate().await?;
}
Ok(())
}
pub(crate) async fn stderr_tail(&self) -> String {
if let Some(stderr) = self.stderr.lock().await.as_ref().cloned() {
return stderr.tail_string().await;
}
String::new()
}
}
pub(crate) struct SandboxAcpTransport {
command: AcpCommand,
cwd: String,
env: HashMap<String, String>,
sandbox: Arc<dyn Sandbox>,
state: TransportState,
}
impl SandboxAcpTransport {
pub(crate) fn new(
command: AcpCommand,
cwd: String,
env: HashMap<String, String>,
sandbox: Arc<dyn Sandbox>,
state: TransportState,
) -> Self {
Self {
command,
cwd,
env,
sandbox,
state,
}
}
}
impl ConnectTo<Client> for SandboxAcpTransport {
async fn connect_to(self, client: impl ConnectTo<Agent>) -> AcpProtocolResult<()> {
let mut env = self.command.env().clone();
env.extend(self.env);
let process = match self
.sandbox
.spawn_stdio_process(
&self.command.to_shell_command(),
Some(&self.cwd),
Some(&env),
None,
)
.await
{
Ok(process) => process,
Err(error) => {
self.state.set_startup_error(error).await;
return Err(internal_error("ACP process failed to start"));
}
};
let handle = process.handle.clone();
let stderr = process.stderr.clone();
self.state.set_process(handle.clone(), stderr.clone()).await;
let incoming_lines = Box::pin(BufReader::new(process.stdout.compat()).lines())
as Pin<Box<dyn Stream<Item = IoResult<String>> + Send>>;
let outgoing_sink = Box::pin(unfold(
process.stdin.compat_write(),
async move |mut writer, line: String| {
let mut bytes = line.into_bytes();
bytes.push(b'\n');
writer.write_all(&bytes).await?;
Ok::<_, std::io::Error>(writer)
},
));
let protocol = agent_client_protocol::ConnectTo::<Client>::connect_to(
Lines::new(outgoing_sink, incoming_lines),
client,
);
tokio::select! {
result = protocol => {
if let Err(err) = handle.terminate().await {
tracing::warn!(error = %err, "Failed to terminate ACP process after protocol completion");
}
let _ = timeout(Duration::from_millis(500), handle.wait()).await;
result
}
termination = handle.wait() => {
let termination = termination.map_err(ProtocolError::into_internal_error)?;
let stderr = stderr.tail_string().await;
let exit_code = termination
.exit_code
.map_or_else(|| "unknown".to_string(), |code| code.to_string());
Err(internal_error(format!(
"ACP process exited before protocol completed: termination={}, exit_code={exit_code}, stderr={stderr}",
termination.termination,
)))
}
}
}
}

View file

@ -0,0 +1,449 @@
use std::collections::HashMap;
use std::path::Path;
use std::sync::Arc;
use std::time::Duration;
use agent_client_protocol::schema::StopReason;
use fabro_acp::{AcpError, AcpRunRequest, AcpRunResult, resolve_acp_command, run_acp_turn};
use fabro_sandbox::test_support::MockSandbox;
use fabro_sandbox::{LocalSandbox, Sandbox, shell_quote};
use fabro_util::error::collect_chain;
use tokio::fs::{read_to_string, write};
use tokio::process::Command;
use tokio::sync::Notify;
use tokio::time::{sleep, timeout};
use tokio_util::sync::CancellationToken;
const ACP_TEST_TIMEOUT_MS: u64 = 30_000;
#[allow(
unused,
unreachable_pub,
reason = "integration test imports the shared test fixture source as a private module"
)]
#[path = "../src/test_support.rs"]
mod test_support;
use test_support::fake_acp_agent_script;
#[tokio::test]
async fn stdio_spawn_failure_returns_sandbox_error() {
const SANDBOX_FAILURE: &str = "ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet";
let command = resolve_acp_command(Some("fake-acp-agent")).expect("resolve ACP command");
let mut sandbox = MockSandbox::linux();
sandbox.stdio_process_error = Some(SANDBOX_FAILURE.to_string());
let sandbox: Arc<dyn Sandbox> = Arc::new(sandbox);
let result = run_acp_turn(AcpRunRequest {
command,
prompt: "hello".to_string(),
cwd: "/workspace".to_string(),
timeout_ms: Some(ACP_TEST_TIMEOUT_MS),
env: HashMap::new(),
sandbox,
cancel_token: CancellationToken::new(),
on_activity: None,
})
.await;
let Err(error) = result else {
panic!("stdio spawn failure should fail");
};
assert!(
matches!(error, AcpError::Sandbox(_)),
"expected sandbox error, got {error:?}"
);
let chain = collect_chain(&error);
assert!(
chain.iter().any(|cause| cause == SANDBOX_FAILURE),
"cause chain should contain sandbox failure, got: {chain:?}"
);
}
#[tokio::test]
async fn session_lifecycle_initializes_sends_prompt_and_aggregates_text() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let script_path = tempdir.path().join("fake_acp_agent.py");
let record_path = tempdir.path().join("methods.txt");
write(&script_path, fake_acp_agent_script())
.await
.expect("write fake ACP agent");
let raw_command = format!("python3 {}", shell_quote(&script_path.to_string_lossy()));
let command = resolve_acp_command(Some(&raw_command)).expect("resolve ACP command");
let sandbox: Arc<dyn Sandbox> = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf()));
let result = run_acp_turn(AcpRunRequest {
command,
prompt: "hello".to_string(),
cwd: tempdir.path().to_string_lossy().into_owned(),
timeout_ms: Some(ACP_TEST_TIMEOUT_MS),
env: HashMap::from([(
"ACP_RECORD".to_string(),
record_path.to_string_lossy().into_owned(),
)]),
sandbox,
cancel_token: CancellationToken::new(),
on_activity: None,
})
.await
.expect("run ACP turn");
assert_eq!(result.text, "hello from acp");
assert_eq!(result.stop_reason, StopReason::EndTurn);
assert_eq!(
read_to_string(record_path)
.await
.expect("read method record"),
"initialize\nsession/new\nsession/prompt\n"
);
}
#[tokio::test]
async fn permission_request_selects_allow_always() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let permission_path = tempdir.path().join("permission.json");
let result = run_fake_agent(
tempdir.path(),
HashMap::from([
("ACP_MODE".to_string(), "permission".to_string()),
(
"ACP_PERMISSION".to_string(),
permission_path.to_string_lossy().into_owned(),
),
]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect("run ACP turn");
assert_eq!(result.text, "hello from acp");
let permission = read_to_string(permission_path)
.await
.expect("read permission record");
assert!(permission.contains(r#""outcome":"selected""#));
assert!(permission.contains(r#""optionId":"always""#));
}
#[tokio::test]
async fn runs_inside_sandbox_and_uses_requested_cwd() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let cwd_path = tempdir.path().join("session_new.json");
let result = run_fake_agent(
tempdir.path(),
HashMap::from([
("ACP_MODE".to_string(), "write_file".to_string()),
(
"ACP_SESSION_NEW_PARAMS".to_string(),
cwd_path.to_string_lossy().into_owned(),
),
]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect("run ACP turn");
assert_eq!(result.text, "hello from acp");
assert_eq!(
read_to_string(tempdir.path().join("hello.txt"))
.await
.expect("read sandbox output file"),
"hello from sandbox\n"
);
assert!(
read_to_string(cwd_path)
.await
.expect("read session/new params")
.contains(&tempdir.path().to_string_lossy().into_owned())
);
}
#[tokio::test]
async fn cancellation_sends_session_cancel_and_returns_cancelled() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let cancel_path = tempdir.path().join("cancel.txt");
let tempdir_path = tempdir.path().to_path_buf();
let cancel_path_for_task = cancel_path.clone();
let cancel_token = CancellationToken::new();
let cancel_for_task = cancel_token.clone();
let prompt_started = Arc::new(Notify::new());
let prompt_started_for_task = prompt_started.clone();
let task = tokio::spawn(async move {
run_fake_agent_with_activity(
&tempdir_path,
HashMap::from([
("ACP_MODE".to_string(), "cancel".to_string()),
(
"ACP_CANCEL_RECORD".to_string(),
cancel_path_for_task.to_string_lossy().into_owned(),
),
]),
Some(ACP_TEST_TIMEOUT_MS),
cancel_for_task,
Some(Arc::new(move || prompt_started_for_task.notify_one())),
)
.await
});
timeout(
Duration::from_millis(ACP_TEST_TIMEOUT_MS),
prompt_started.notified(),
)
.await
.expect("fake ACP agent should acknowledge session/prompt before cancellation");
cancel_token.cancel();
let err = task
.await
.expect("join cancellation task")
.expect_err("cancelled turn should error");
assert!(matches!(err, AcpError::Cancelled));
assert_eq!(
read_to_string(cancel_path)
.await
.expect("read cancel record"),
"session/cancel\n"
);
}
#[tokio::test]
async fn pre_session_cancellation_returns_cancelled() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let cancel_token = CancellationToken::new();
cancel_token.cancel();
let err = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_MODE".to_string(), "slow_initialize".to_string())]),
Some(1_000),
cancel_token,
)
.await
.expect_err("pre-session cancellation should error");
assert!(matches!(err, AcpError::Cancelled));
}
#[tokio::test]
async fn successful_turn_terminates_lingering_agent_process() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let pid_path = tempdir.path().join("agent.pid");
let result = run_fake_agent(
tempdir.path(),
HashMap::from([
("ACP_MODE".to_string(), "linger_after_response".to_string()),
(
"ACP_PID_RECORD".to_string(),
pid_path.to_string_lossy().into_owned(),
),
]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect("run ACP turn");
sleep(Duration::from_millis(100)).await;
let pid = read_to_string(&pid_path).await.expect("read agent pid");
let still_running = process_is_running(pid.trim()).await;
if still_running {
let _ = Command::new("kill")
.arg("-TERM")
.arg(pid.trim())
.status()
.await;
}
assert_eq!(result.text, "hello from acp");
assert!(
!still_running,
"successful ACP turn should not leave lingering agent process"
);
}
#[tokio::test]
async fn refusal_stop_reason_returns_text() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let result = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_STOP_REASON".to_string(), "refusal".to_string())]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect("run ACP turn");
assert_eq!(result.text, "hello from acp");
assert_eq!(result.stop_reason, StopReason::Refusal);
}
#[tokio::test]
async fn max_tokens_stop_reason_returns_partial_text_error() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let err = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_STOP_REASON".to_string(), "max_tokens".to_string())]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect_err("max_tokens should return stop reason error");
let AcpError::StopReason { stop_reason, text } = err else {
panic!("expected stop reason error");
};
assert_eq!(stop_reason, "max_tokens");
assert_eq!(text, "hello from acp");
}
#[tokio::test]
async fn max_turn_requests_stop_reason_returns_partial_text_error() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let err = run_fake_agent(
tempdir.path(),
HashMap::from([(
"ACP_STOP_REASON".to_string(),
"max_turn_requests".to_string(),
)]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect_err("max_turn_requests should return stop reason error");
let AcpError::StopReason { stop_reason, text } = err else {
panic!("expected stop reason error");
};
assert_eq!(stop_reason, "max_turn_requests");
assert_eq!(text, "hello from acp");
}
#[tokio::test]
async fn timeout_terminates_process_and_returns_timeout() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let err = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_MODE".to_string(), "timeout".to_string())]),
Some(100),
CancellationToken::new(),
)
.await
.expect_err("timeout should error");
assert!(matches!(err, AcpError::TimedOut { .. }));
}
#[tokio::test]
async fn malformed_json_returns_protocol_error() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let err = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_MODE".to_string(), "malformed".to_string())]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect_err("malformed JSON should error");
assert!(matches!(err, AcpError::Protocol(_)));
}
#[tokio::test]
async fn early_exit_returns_protocol_error_with_stderr() {
let tempdir = tempfile::tempdir().expect("create tempdir");
let err = run_fake_agent(
tempdir.path(),
HashMap::from([("ACP_MODE".to_string(), "early_exit".to_string())]),
Some(ACP_TEST_TIMEOUT_MS),
CancellationToken::new(),
)
.await
.expect_err("early exit should error");
let AcpError::Protocol(error) = err else {
panic!("expected protocol error");
};
let message = error.to_string();
assert!(
message.contains("exit_code=2"),
"early exit should include exit code in diagnostic: {message}"
);
assert!(
message.contains("early boom"),
"early exit should include stderr tail in diagnostic: {message}"
);
}
async fn run_fake_agent(
tempdir: &Path,
env: HashMap<String, String>,
timeout_ms: Option<u64>,
cancel_token: CancellationToken,
) -> Result<AcpRunResult, AcpError> {
run_fake_agent_with_activity(tempdir, env, timeout_ms, cancel_token, None).await
}
async fn run_fake_agent_with_activity(
tempdir: &Path,
env: HashMap<String, String>,
timeout_ms: Option<u64>,
cancel_token: CancellationToken,
on_activity: Option<Arc<dyn Fn() + Send + Sync>>,
) -> Result<AcpRunResult, AcpError> {
let script_path = tempdir.join("fake_acp_agent.py");
write(&script_path, fake_acp_agent_script())
.await
.expect("write fake ACP agent");
let raw_command = format!("python3 {}", shell_quote(&script_path.to_string_lossy()));
let command = resolve_acp_command(Some(&raw_command)).expect("resolve ACP command");
let sandbox: Arc<dyn Sandbox> = Arc::new(LocalSandbox::new(tempdir.to_path_buf()));
run_acp_turn(AcpRunRequest {
command,
prompt: "hello".to_string(),
cwd: tempdir.to_string_lossy().into_owned(),
timeout_ms,
env,
sandbox,
cancel_token,
on_activity,
})
.await
}
async fn process_is_running(pid: &str) -> bool {
let Ok(status) = Command::new("kill").arg("-0").arg(pid).status().await else {
return false;
};
if !status.success() {
return false;
}
let Ok(output) = Command::new("ps")
.args(["-ww", "-o", "stat=", "-p", pid])
.output()
.await
else {
return true;
};
if !output.status.success() {
return false;
}
String::from_utf8_lossy(&output.stdout)
.chars()
.find(|ch| !ch.is_whitespace())
.is_none_or(|state| !matches!(state, 'Z' | 'z'))
}

View file

@ -41,8 +41,9 @@ pub use profiles::{AnthropicProfile, EnvContext, GeminiProfile, OpenAiProfile};
pub use read_before_write_sandbox::ReadBeforeWriteSandbox;
pub use sandbox::{
CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox,
SandboxEvent, SandboxEventCallback, WorktreeEvent, WorktreeEventCallback, WorktreeOptions,
WorktreeSandbox, format_lines_numbered, shell_quote,
SandboxEvent, SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle,
WorktreeEvent, WorktreeEventCallback, WorktreeOptions, WorktreeSandbox, format_lines_numbered,
shell_quote,
};
pub use session::{
CompletionCoordinator, Session, SessionControlHandle, StaticEnvProvider, SteeringItem,

View file

@ -62,6 +62,8 @@ mod tests {
command: vec!["python3".into(), test_server],
env: HashMap::new(),
},
current_dir: None,
clear_env: false,
startup_timeout_secs: 10,
tool_timeout_secs: 30,
}

View file

@ -3,6 +3,7 @@
// `crate::delegate_sandbox!` invocations continue to work.
pub use fabro_sandbox::{
CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox,
SandboxEvent, SandboxEventCallback, WorktreeEvent, WorktreeEventCallback, WorktreeOptions,
SandboxEvent, SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle,
StdioProcessTermination, WorktreeEvent, WorktreeEventCallback, WorktreeOptions,
WorktreeSandbox, delegate_sandbox, format_lines_numbered, shell_quote,
};

View file

@ -493,6 +493,8 @@ impl Session {
resolved.push(McpServerSettings {
name: config.name.clone(),
transport: McpTransport::Http { url, headers },
current_dir: config.current_dir.clone(),
clear_env: config.clear_env,
startup_timeout_secs: config.startup_timeout_secs,
tool_timeout_secs: config.tool_timeout_secs,
});
@ -3367,6 +3369,8 @@ mod tests {
command: vec!["python3".into(), test_server],
env: HashMap::new(),
},
current_dir: None,
clear_env: false,
startup_timeout_secs: 10,
tool_timeout_secs: 30,
}],

View file

@ -31,6 +31,8 @@ fabro-hooks = { path = "../fabro-hooks" }
fabro-install = { path = "../fabro-install" }
fabro-interview = { path = "../fabro-interview" }
fabro-mcp = { path = "../fabro-mcp" }
fabro-mcp-server = { path = "../fabro-mcp-server" }
fabro-manifest = { path = "../fabro-manifest" }
fabro-proc = { path = "../fabro-proc" }
fabro-sandbox = { path = "../fabro-sandbox", features = ["daytona"] }
fabro-checkpoint = { path = "../fabro-checkpoint" }
@ -113,6 +115,7 @@ chrono = { workspace = true }
[dev-dependencies]
assert_cmd = "2"
fabro-acp = { path = "../fabro-acp", features = ["test-support"] }
fabro-build-support = { path = "../build-support" }
fabro-server = { path = "../fabro-server", features = ["test-support"] }
insta = { workspace = true, features = ["filters"] }

View file

@ -168,6 +168,49 @@ pub(crate) struct ServerConnectionArgs {
pub(crate) target: ServerTargetArgs,
}
#[derive(Args)]
pub(crate) struct McpNamespace {
#[command(subcommand)]
pub(crate) command: McpCommand,
}
#[derive(Subcommand)]
pub(crate) enum McpCommand {
/// Start the Fabro MCP server over stdio
Start(McpStartArgs),
/// Print MCP client configuration JSON
Config(McpConfigArgs),
/// Configure an MCP client to launch Fabro
Init(McpInitArgs),
}
#[derive(Args, Debug, Clone, Default)]
pub(crate) struct McpStartArgs {
#[command(flatten)]
pub(crate) connection: ServerConnectionArgs,
}
#[derive(Args, Debug, Clone, Default)]
pub(crate) struct McpConfigArgs {
#[command(flatten)]
pub(crate) connection: ServerConnectionArgs,
}
#[derive(Args, Debug, Clone)]
pub(crate) struct McpInitArgs {
pub(crate) agent: McpAgent,
#[command(flatten)]
pub(crate) connection: ServerConnectionArgs,
}
#[derive(Debug, Clone, Copy, ValueEnum)]
pub(crate) enum McpAgent {
Claude,
Cursor,
Windsurf,
}
#[derive(Args, Debug, Clone, Default)]
pub(crate) struct InputOverrideArgs {
/// Override a workflow input value (repeatable, format: KEY=VALUE)
@ -1118,6 +1161,8 @@ pub(crate) enum Commands {
#[command(subcommand)]
command: Option<ModelsCommand>,
},
/// Model Context Protocol server
Mcp(McpNamespace),
/// Server operations
Server(ServerNamespace),
/// Check environment and integration health
@ -1209,6 +1254,11 @@ impl Commands {
Some(ModelsCommand::Test(_)) => "model test",
None => "model",
},
Self::Mcp(ns) => match &ns.command {
McpCommand::Start(_) => "mcp start",
McpCommand::Config(_) => "mcp config",
McpCommand::Init(_) => "mcp init",
},
Self::Server(ns) => match &ns.command {
ServerCommand::Start(_) => "server start",
ServerCommand::Stop(_) => "server stop",

View file

@ -102,6 +102,10 @@ impl CommandContext {
&self.cwd
}
pub(crate) fn storage_dir(&self) -> &Path {
&self.storage_dir
}
pub(crate) fn run_settings(&self) -> Result<&RunNamespace> {
self.run_settings
.as_ref()

View file

@ -12,13 +12,13 @@ use std::io::Write;
use anyhow::{Context, bail};
use fabro_api::types;
use fabro_config::user::active_settings_path;
use fabro_manifest::{ManifestBuildInput, build_run_manifest};
use fabro_util::terminal::Styles;
use tracing::debug;
use crate::args::{GraphArgs, GraphDirection, GraphOutputFormat};
use crate::command_context::CommandContext;
use crate::commands::run::output::api_diagnostics_to_local;
use crate::manifest_builder::{ManifestBuildInput, build_run_manifest};
use crate::shared::{absolute_or_current, print_diagnostics, print_json_pretty, relative_path};
pub(crate) async fn run(

View file

@ -0,0 +1,91 @@
use std::fmt::Write as _;
use anyhow::{Context as _, Result};
use crate::args::{McpAgent, McpCommand, McpNamespace, ServerConnectionArgs};
use crate::command_context::CommandContext;
use crate::server_client;
pub(crate) async fn dispatch(ns: McpNamespace, base_ctx: &CommandContext) -> Result<()> {
match ns.command {
McpCommand::Start(args) => {
fabro_mcp_server::start(server_settings(base_ctx, &args.connection)?).await
}
McpCommand::Config(args) => {
let json = fabro_mcp_server::config_json(&config_settings(&args.connection))?;
let _ = write!(base_ctx.printer().stdout_important(), "{json}");
Ok(())
}
McpCommand::Init(args) => {
fabro_mcp_server::init_agent(&init_settings(args.agent, &args.connection)?)?;
Ok(())
}
}
}
fn server_settings(
base_ctx: &CommandContext,
connection: &ServerConnectionArgs,
) -> Result<fabro_mcp_server::FabroMcpServerSettings> {
let connection_ctx = base_ctx.with_connection(connection)?;
let target = connection.target.clone();
let user_settings = connection_ctx.user_settings().clone();
let storage_dir = connection_ctx.storage_dir().to_path_buf();
let base_config_path = connection_ctx.base_config_path().to_path_buf();
let config_path = base_config_path.clone();
let client_factory: fabro_mcp_server::FabroClientFactory = std::sync::Arc::new(move || {
let target = target.clone();
let user_settings = user_settings.clone();
let storage_dir = storage_dir.clone();
let base_config_path = base_config_path.clone();
let future: fabro_mcp_server::FabroClientFuture = Box::pin(async move {
server_client::connect_server_with_settings(
&target,
&user_settings,
&storage_dir,
&base_config_path,
)
.await
});
future
});
Ok(fabro_mcp_server::FabroMcpServerSettings {
client_factory,
config_path,
cwd: base_ctx.cwd().to_path_buf(),
})
}
fn init_settings(
agent: McpAgent,
connection: &ServerConnectionArgs,
) -> Result<fabro_mcp_server::McpInitSettings> {
Ok(fabro_mcp_server::McpInitSettings {
agent: McpAgentForServer(agent).into(),
config: config_settings(connection),
home_dir: home_dir()?,
})
}
fn config_settings(connection: &ServerConnectionArgs) -> fabro_mcp_server::McpConfigSettings {
fabro_mcp_server::McpConfigSettings {
server: connection.target.server.clone(),
storage_dir: connection.storage_dir.clone_path(),
}
}
fn home_dir() -> Result<std::path::PathBuf> {
dirs::home_dir().context("failed to resolve home directory for MCP config")
}
struct McpAgentForServer(McpAgent);
impl From<McpAgentForServer> for fabro_mcp_server::McpAgent {
fn from(value: McpAgentForServer) -> Self {
match value.0 {
McpAgent::Claude => Self::Claude,
McpAgent::Cursor => Self::Cursor,
McpAgent::Windsurf => Self::Windsurf,
}
}
}

View file

@ -7,6 +7,7 @@ pub(crate) mod dump;
pub(crate) mod exec;
pub(crate) mod graph;
pub(crate) mod install;
pub(crate) mod mcp;
pub(crate) mod model;
pub(crate) mod parse;
pub(crate) mod pr;

View file

@ -1,5 +1,6 @@
use anyhow::bail;
use fabro_config::user::active_settings_path;
use fabro_manifest::{ManifestBuildInput, build_run_manifest};
use fabro_util::terminal::Styles;
use crate::args::PreflightArgs;
@ -8,7 +9,7 @@ use crate::commands::run::output::{
api_check_report_to_local, api_diagnostics_to_local, print_workflow_summary,
};
use crate::commands::run::overrides::preflight_args_overrides;
use crate::manifest_builder::{ManifestBuildInput, build_run_manifest, preflight_manifest_args};
use crate::manifest_args::preflight_manifest_args;
use crate::shared::{cyan_spinner, print_json_pretty};
pub(crate) async fn execute(

View file

@ -1,6 +1,7 @@
use anyhow::{Context as _, bail};
use fabro_config::RunLayer;
use fabro_config::user::active_settings_path;
use fabro_manifest::{ManifestBuildInput, build_run_manifest};
use fabro_server::manifest_validation;
use fabro_types::RunId;
use fabro_util::terminal::Styles;
@ -9,7 +10,7 @@ use super::output::{api_diagnostics_to_local, print_workflow_summary};
use super::overrides::run_args_overrides;
use crate::args::RunArgs;
use crate::command_context::CommandContext;
use crate::manifest_builder::{ManifestBuildInput, build_run_manifest, run_manifest_args};
use crate::manifest_args::run_manifest_args;
pub(crate) struct CreatedRun {
pub(crate) run_id: RunId,

View file

@ -2,14 +2,11 @@ use std::collections::HashMap;
use std::path::{Path, PathBuf};
use anyhow::{Result, anyhow};
use fabro_config::{
CliLayer, CliOutputLayer, ReplaceMap, RunExecutionLayer, RunGoalLayer, RunLayer, RunModelLayer,
RunSandboxLayer, parse_input_overrides,
};
use fabro_config::{CliLayer, CliOutputLayer, RunGoalLayer, RunLayer, parse_input_overrides};
use fabro_manifest::{RunOverrideInput, build_run_overrides};
use fabro_sandbox::SandboxProvider;
use fabro_types::settings::cli::OutputVerbosity;
use fabro_types::settings::interp::InterpString;
use fabro_types::settings::run::{ApprovalMode, RunMode};
use crate::args::{PreflightArgs, RunArgs};
@ -32,48 +29,6 @@ pub(crate) fn parse_labels(labels: &[String]) -> HashMap<String, String> {
.collect()
}
fn model_from_args(model: Option<&str>, provider: Option<&str>) -> Option<RunModelLayer> {
if model.is_none() && provider.is_none() {
return None;
}
Some(RunModelLayer {
provider: provider.map(InterpString::parse),
name: model.map(InterpString::parse),
fallbacks: Vec::new(),
controls: None,
})
}
fn sandbox_layer(
sandbox: Option<SandboxProvider>,
preserve: Option<bool>,
) -> Option<RunSandboxLayer> {
if sandbox.is_none() && preserve.is_none() {
return None;
}
Some(RunSandboxLayer {
provider: sandbox.map(|p| p.to_string()),
preserve,
..RunSandboxLayer::default()
})
}
fn execution_layer(dry_run: Option<bool>, auto_approve: Option<bool>) -> Option<RunExecutionLayer> {
if dry_run.is_none() && auto_approve.is_none() {
return None;
}
Some(RunExecutionLayer {
mode: dry_run.map(|d| if d { RunMode::DryRun } else { RunMode::Normal }),
approval: auto_approve.map(|a| {
if a {
ApprovalMode::Auto
} else {
ApprovalMode::Prompt
}
}),
})
}
fn cli_layer_for_verbose(verbose: bool) -> Option<CliLayer> {
verbose.then(|| CliLayer {
output: Some(CliOutputLayer {
@ -120,24 +75,22 @@ fn current_dir_or_dot() -> PathBuf {
}
pub(crate) fn run_args_overrides(args: &RunArgs) -> Result<ManifestSettingsOverrides> {
let model = model_from_args(args.model.as_deref(), args.provider.as_deref());
let sandbox = sandbox_layer(
args.sandbox.map(Into::into),
sparse_flag(args.preserve_sandbox),
);
let execution = execution_layer(sparse_flag(args.dry_run), sparse_flag(args.auto_approve));
let cwd = current_dir_or_dot();
let goal = goal_layer_from_args(args.goal.as_deref(), args.goal_file.as_deref(), &cwd)?;
let run = RunLayer {
goal,
metadata: ReplaceMap::from(parse_labels(&args.label)),
model,
sandbox,
execution,
..RunLayer::default()
};
let sandbox = args.sandbox.map(SandboxProvider::from);
let sandbox_provider = sandbox.as_ref().map(ToString::to_string);
let mut run = build_run_overrides(RunOverrideInput {
goal: None,
model: args.model.as_deref(),
provider: args.provider.as_deref(),
sandbox: sandbox_provider.as_deref(),
docker_image: None,
preserve_sandbox: sparse_flag(args.preserve_sandbox),
dry_run: sparse_flag(args.dry_run),
auto_approve: sparse_flag(args.auto_approve),
labels: parse_labels(&args.label),
});
run.goal = goal;
Ok(ManifestSettingsOverrides {
run: Some(run),
@ -147,21 +100,23 @@ pub(crate) fn run_args_overrides(args: &RunArgs) -> Result<ManifestSettingsOverr
}
pub(crate) fn preflight_args_overrides(args: &PreflightArgs) -> Result<ManifestSettingsOverrides> {
let model = model_from_args(args.model.as_deref(), args.provider.as_deref());
let sandbox = args.sandbox.map(|s| RunSandboxLayer {
provider: Some(SandboxProvider::from(s).to_string()),
..RunSandboxLayer::default()
});
let cwd = current_dir_or_dot();
let goal = goal_layer_from_args(args.goal.as_deref(), args.goal_file.as_deref(), &cwd)?;
let run = RunLayer {
goal,
model,
sandbox,
..RunLayer::default()
};
let sandbox_provider = args
.sandbox
.map(|sandbox| SandboxProvider::from(sandbox).to_string());
let mut run = build_run_overrides(RunOverrideInput {
goal: None,
model: args.model.as_deref(),
provider: args.provider.as_deref(),
sandbox: sandbox_provider.as_deref(),
docker_image: None,
preserve_sandbox: None,
dry_run: None,
auto_approve: None,
labels: HashMap::new(),
});
run.goal = goal;
Ok(ManifestSettingsOverrides {
run: Some(run),

View file

@ -472,6 +472,7 @@ mod tests {
use fabro_agent::{AgentEvent, SandboxEvent};
use fabro_llm::types::TokenCounts;
use fabro_model::{ModelRef, Provider};
use fabro_types::run_event::CliEnsureCompletedProps;
use fabro_types::{
MetadataSnapshotFailureKind, MetadataSnapshotPhase, ParallelBranchId, SandboxProvider,
StageId, fixtures,
@ -527,6 +528,24 @@ mod tests {
ui.handle_event(&stored);
}
fn emit_body(ui: &mut ProgressUI, body: fabro_types::EventBody) {
ui.handle_event(&RunEvent {
id: "evt_legacy".to_string(),
ts: Utc::now(),
run_id: fixtures::RUN_1,
node_id: None,
node_label: None,
stage_id: None,
parallel_group_id: None,
parallel_branch_id: None,
session_id: None,
parent_session_id: None,
tool_call_id: None,
actor: None,
body,
});
}
fn agent_event(stage: &str, event: AgentEvent) -> Event {
Event::Agent {
stage: stage.into(),
@ -860,13 +879,16 @@ mod tests {
});
emit(&mut ui, Event::SetupStarted { command_count: 2 });
emit(&mut ui, Event::SetupCompleted { duration_ms: 8200 });
emit(&mut ui, Event::CliEnsureCompleted {
cli_name: "gh".into(),
provider: "github".into(),
already_installed: false,
node_installed: false,
duration_ms: 600,
});
emit_body(
&mut ui,
fabro_types::EventBody::CliEnsureCompleted(CliEnsureCompletedProps {
cli_name: "gh".into(),
provider: "github".into(),
already_installed: false,
node_installed: false,
duration_ms: 600,
}),
);
emit(&mut ui, Event::DevcontainerResolved {
dockerfile_lines: 24,
environment_count: 3,

View file

@ -8,5 +8,5 @@ pub(crate) async fn start_run_with_client(
run_id: &RunId,
resume: bool,
) -> Result<()> {
client.start_run(run_id, resume).await
client.start_run(run_id, resume).await.map(|_| ())
}

View file

@ -71,7 +71,7 @@ async fn run_bulk(action: Action, identifiers: &[String], ctx: &CommandContext)
Action::Unarchive => client.unarchive_run(&run_id).await,
};
match result {
Ok(()) => {
Ok(_) => {
let run_id_string = run_id.to_string();
changed.push(run_id_string.clone());
if !json {

View file

@ -1,13 +1,13 @@
use anyhow::bail;
use fabro_config::RunLayer;
use fabro_config::user::active_settings_path;
use fabro_manifest::{ManifestBuildInput, build_run_manifest};
use fabro_server::manifest_validation;
use fabro_util::terminal::Styles;
use crate::args::ValidateArgs;
use crate::command_context::CommandContext;
use crate::commands::run::output::api_diagnostics_to_local;
use crate::manifest_builder::{ManifestBuildInput, build_run_manifest};
use crate::shared::{print_diagnostics, print_json_pretty, relative_path};
pub(crate) fn run(

View file

@ -1,9 +0,0 @@
#![expect(
dead_code,
reason = "the library exports manifest builder helpers while the binary owns most CLI dispatch"
)]
mod args;
mod manifest_builder;
pub use manifest_builder::{BuiltManifest, ManifestBuildInput, build_run_manifest};

View file

@ -10,11 +10,7 @@ mod gh;
mod landing;
mod local_server;
mod logging;
#[allow(
unreachable_pub,
reason = "The library exports manifest builder helpers for tests; the binary includes the same module privately."
)]
mod manifest_builder;
mod manifest_args;
mod server_client;
mod server_runs;
mod shared;
@ -281,6 +277,9 @@ async fn main_inner(worker_token: Option<String>) -> (String, Result<()>) {
Commands::Model { command } => {
commands::model::execute(command, &base_ctx).await?;
}
Commands::Mcp(ns) => {
commands::mcp::dispatch(ns, &base_ctx).await?;
}
Commands::Server(ns) => {
Box::pin(commands::server::dispatch(
ns.command,
@ -1196,7 +1195,7 @@ destination = "{destination}"
.expect("should parse");
match *cli.command.unwrap() {
Commands::RunCmd(RunCommands::Run(args)) => {
let manifest_args = manifest_builder::run_manifest_args(&args)
let manifest_args = manifest_args::run_manifest_args(&args)
.expect("input-only args should be retained");
assert_eq!(manifest_args.input, vec!["foo=bar"]);
}

View file

@ -0,0 +1,39 @@
use fabro_api::types;
use crate::args::{PreflightArgs, RunArgs};
pub(crate) fn run_manifest_args(args: &RunArgs) -> Option<types::ManifestArgs> {
let payload = types::ManifestArgs {
auto_approve: args.auto_approve.then_some(true),
dry_run: args.dry_run.then_some(true),
label: args.label.clone(),
model: args.model.clone(),
preserve_sandbox: args.preserve_sandbox.then_some(true),
provider: args.provider.clone(),
sandbox: args
.sandbox
.map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()),
docker_image: None,
input: args.inputs.values.clone(),
verbose: args.verbose.then_some(true),
};
(!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload)
}
pub(crate) fn preflight_manifest_args(args: &PreflightArgs) -> Option<types::ManifestArgs> {
let payload = types::ManifestArgs {
auto_approve: None,
dry_run: None,
label: Vec::new(),
model: args.model.clone(),
preserve_sandbox: None,
provider: args.provider.clone(),
sandbox: args
.sandbox
.map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()),
docker_image: None,
input: args.inputs.values.clone(),
verbose: args.verbose.then_some(true),
};
(!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload)
}

View file

@ -127,18 +127,7 @@ pub(crate) fn color_if(use_color: bool, color: Color) -> Option<Color> {
}
pub(crate) fn run_status_kind(status: RunStatus) -> &'static str {
match status {
RunStatus::Submitted => "submitted",
RunStatus::Queued => "queued",
RunStatus::Starting => "starting",
RunStatus::Running => "running",
RunStatus::Blocked { .. } => "blocked",
RunStatus::Paused { .. } => "paused",
RunStatus::Removing => "removing",
RunStatus::Succeeded { .. } => "succeeded",
RunStatus::Failed { .. } => "failed",
RunStatus::Dead => "dead",
}
status.kind().into()
}
pub(crate) fn split_run_path(s: &str) -> Option<(&str, &str)> {

View file

@ -5,7 +5,11 @@
use std::process::Output;
use fabro_auth::{AuthCredential, AuthDetails};
use fabro_config::Storage;
use fabro_model::Provider;
use fabro_test::{fabro_snapshot, test_context, twin_openai};
use fabro_vault::{SecretType, Vault};
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
@ -13,6 +17,35 @@ async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
.expect("blocking command task should complete")
}
fn toml_path(path: &std::path::Path) -> String {
path.display()
.to_string()
.replace('\\', "\\\\")
.replace('"', "\\\"")
}
fn seed_openai_vault(storage_dir: &std::path::Path, base_url: &str, api_key: &str) {
let mut vault =
Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load");
vault
.set(
"openai",
&serde_json::to_string(&AuthCredential {
provider: Provider::OpenAi,
details: AuthDetails::ApiKey {
key: api_key.to_string(),
},
})
.expect("OpenAI test credential should serialize"),
SecretType::Credential,
None,
)
.expect("OpenAI credential should store in test vault");
vault
.set("OPENAI_BASE_URL", base_url, SecretType::Environment, None)
.expect("OpenAI base URL should store in test vault");
}
#[test]
fn help() {
let context = test_context!();
@ -64,9 +97,28 @@ fn live_doctor() {
#[fabro_macros::e2e_test(twin)]
async fn twin_doctor() {
let context = test_context!();
let mut context = test_context!();
let twin = twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
let storage_dir = context.temp_dir.join("doctor-server-storage");
context.write_home(
".fabro/settings.toml",
format!(
r#"[server.storage]
root = "{}"
[server.auth]
methods = ["dev-token"]
[server.integrations.github]
strategy = "app"
"#,
toml_path(&storage_dir)
),
);
seed_openai_vault(&storage_dir, &twin.base_url, &namespace);
context.isolated_server();
let mut cmd = context.doctor();
cmd.arg("--verbose");
cmd.env_clear();

View file

@ -33,6 +33,7 @@ fn help() {
archive Mark terminal runs as archived (reviewed, no further action needed). Archived runs are hidden from default listings
unarchive Restore archived runs to their prior terminal status
model List and test LLM models
mcp Model Context Protocol server
server Server operations
doctor Check environment and integration health
version Show client and server version information

File diff suppressed because it is too large Load diff

View file

@ -20,6 +20,7 @@ mod inspect;
mod install;
mod json_global;
mod logs;
mod mcp;
mod model;
mod model_list;
mod model_test;

View file

@ -26,14 +26,17 @@ fn local_run_lifecycle() {
};
// 1. Run a workflow
cmd(&[
"run",
"--auto-approve",
"--sandbox",
"local",
fixture("command_pipeline.fabro").to_str().unwrap(),
])
.success();
context
.run_cmd()
.args([
"--auto-approve",
"--sandbox",
"local",
fixture("command_pipeline.fabro").to_str().unwrap(),
])
.timeout(timeout_for("local"))
.assert()
.success();
// 2. ps -a --json — should list exactly one run
let label = context.test_case_label();

View file

@ -0,0 +1,208 @@
#![expect(
clippy::disallowed_methods,
reason = "integration test initializes an isolated git repository with the system git binary"
)]
use fabro_acp::test_support::fake_acp_agent_script;
use fabro_auth::{AuthCredential, AuthDetails};
use fabro_config::Storage;
use fabro_model::Provider;
use fabro_test::test_context;
use fabro_types::EventBody;
use fabro_vault::{SecretType, Vault};
use super::{find_run_dir, has_event, read_conclusion, run_events, run_state};
#[test]
fn acp_backend_workflow() {
let mut context = test_context!();
context.write_home(
".fabro/settings.toml",
"[server.auth]\nmethods = [\"dev-token\"]\n",
);
context.isolated_server();
seed_openai_vault(&context.storage_dir);
let fake_agent = write_fake_acp_agent(&context);
let acp_command = fake_acp_command_attr(&fake_agent);
let workflow = context.temp_dir.join("acp_backend.fabro");
context.write_temp(
"acp_backend.fabro",
format!(
r#"digraph ACP {{
graph [goal="Exercise ACP backend"]
start [shape=Mdiamond]
work [type="agent", backend="acp", provider="openai", model="fake-acp", prompt="write hello.txt", acp_command={acp_command}]
exit [shape=Msquare]
start -> work
work -> exit
}}"#
),
);
init_git_repo(&context.temp_dir);
context
.run_cmd()
.args(["--auto-approve", "--sandbox", "local"])
.arg(&workflow)
.assert()
.success();
let run_dir = find_run_dir(&context);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("succeeded"));
let events = run_events(&run_dir);
assert!(has_event(&run_dir, "agent.acp.started"));
assert!(has_event(&run_dir, "agent.acp.completed"));
let completed = events
.iter()
.find_map(|event| match &event.event.body {
EventBody::StageCompleted(props) if event.event.node_id.as_deref() == Some("work") => {
Some(props)
}
_ => None,
})
.expect("work stage should complete");
assert_eq!(completed.response.as_deref(), Some("hello from acp"));
assert!(
completed
.files_touched
.iter()
.any(|file| file == "hello.txt"),
"files_touched should include hello.txt: {:?}",
completed.files_touched
);
let state = serde_json::to_value(run_state(&run_dir)).expect("run state should serialize");
let stages = state["stages"]
.as_object()
.expect("run state should contain stages");
assert!(
stages.values().any(|stage| {
stage["provider_used"]["mode"] == "acp"
&& stage["provider_used"]["provider"] == "openai"
}),
"run projection should include ACP provider metadata: {stages:?}"
);
}
#[test]
fn acp_prompt_workflow_uses_acp_backend() {
let mut context = test_context!();
context.write_home(
".fabro/settings.toml",
"[server.auth]\nmethods = [\"dev-token\"]\n",
);
context.isolated_server();
seed_openai_vault(&context.storage_dir);
let fake_agent = write_fake_acp_agent(&context);
let acp_command = fake_acp_command_attr(&fake_agent);
let workflow = context.temp_dir.join("acp_prompt_backend.fabro");
context.write_temp(
"acp_prompt_backend.fabro",
format!(
r#"digraph ACP {{
graph [goal="Exercise ACP prompt backend"]
start [shape=Mdiamond]
prompt [type="prompt", backend="acp", provider="openai", model="fake-acp", project_memory=false, prompt="write hello.txt", acp_command={acp_command}]
exit [shape=Msquare]
start -> prompt
prompt -> exit
}}"#
),
);
init_git_repo(&context.temp_dir);
context
.run_cmd()
.args(["--auto-approve", "--sandbox", "local"])
.arg(&workflow)
.assert()
.success();
let run_dir = find_run_dir(&context);
let conclusion = read_conclusion(&run_dir);
assert_eq!(conclusion["status"].as_str(), Some("succeeded"));
let events = run_events(&run_dir);
assert!(has_event(&run_dir, "agent.acp.started"));
assert!(has_event(&run_dir, "agent.acp.completed"));
assert!(
!has_event(&run_dir, "agent.session.activated"),
"ACP prompt should not activate an API-mode agent session"
);
let completed = events
.iter()
.find_map(|event| match &event.event.body {
EventBody::StageCompleted(props)
if event.event.node_id.as_deref() == Some("prompt") =>
{
Some(props)
}
_ => None,
})
.expect("prompt stage should complete");
assert_eq!(completed.response.as_deref(), Some("hello from acp"));
let state = serde_json::to_value(run_state(&run_dir)).expect("run state should serialize");
let stages = state["stages"]
.as_object()
.expect("run state should contain stages");
assert!(
stages.values().any(|stage| {
stage["provider_used"]["mode"] == "acp"
&& stage["provider_used"]["provider"] == "openai"
}),
"run projection should include ACP provider metadata: {stages:?}"
);
}
fn seed_openai_vault(storage_dir: &std::path::Path) {
let mut vault =
Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load");
vault
.set(
"openai",
&serde_json::to_string(&AuthCredential {
provider: Provider::OpenAi,
details: AuthDetails::ApiKey {
key: "test-openai-key".to_string(),
},
})
.expect("OpenAI test credential should serialize"),
SecretType::Credential,
None,
)
.expect("OpenAI credential should store in test vault");
}
fn write_fake_acp_agent(context: &fabro_test::TestContext) -> std::path::PathBuf {
context.write_temp("fake_acp_agent.py", fake_acp_agent_script());
context.temp_dir.join("fake_acp_agent.py")
}
fn fake_acp_command_attr(script_path: &std::path::Path) -> String {
let command = serde_json::json!({
"type": "stdio",
"name": "fake",
"command": "python3",
"args": [script_path.to_string_lossy()],
"env": [{"name": "ACP_MODE", "value": "write_file"}],
})
.to_string();
format!("{command:?}")
}
fn init_git_repo(dir: &std::path::Path) {
let output = std::process::Command::new("git")
.args(["init", "-q"])
.current_dir(dir)
.output()
.expect("git init should run");
assert!(
output.status.success(),
"git init failed\nstdout:\n{}\nstderr:\n{}",
String::from_utf8_lossy(&output.stdout),
String::from_utf8_lossy(&output.stderr)
);
}

View file

@ -49,8 +49,8 @@ fn scenario_command_agent_mixed(sandbox: &str) {
let export_dir = dump_export(&context, &run_id_for(&run_dir));
let stdout =
std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("stdout.log"))
.expect("verify stdout.log should exist");
std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("output.log"))
.expect("verify output.log should exist");
assert!(
stdout.contains("SCENARIO_FLAG_42"),
"verify stdout should contain SCENARIO_FLAG_42, got: {stdout}"

View file

@ -49,8 +49,8 @@ fn scenario_command_pipeline(sandbox: &str) {
let export_dir = dump_export(&context, &run_id_for(&run_dir));
let stdout1 =
std::fs::read_to_string(stage_dump_dir(&export_dir, "step1@1").join("stdout.log"))
.expect("step1 stdout.log should exist");
std::fs::read_to_string(stage_dump_dir(&export_dir, "step1@1").join("output.log"))
.expect("step1 output.log should exist");
assert!(
stdout1.contains("hello-from-step1"),
"step1 stdout should contain hello-from-step1, got: {stdout1}"

View file

@ -74,8 +74,8 @@ fn scenario_full_stack(sandbox: &str) {
// Verify node stdout should contain PASS
let export_dir = dump_export(&context, &run_id_for(&run_dir));
let stdout =
std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("stdout.log"))
.expect("verify stdout.log should exist");
std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("output.log"))
.expect("verify output.log should exist");
assert!(
stdout.contains("PASS"),
"verify stdout should contain PASS, got: {stdout}"

View file

@ -11,9 +11,15 @@
use std::process::Output;
use fabro_test::{TestMode, TwinScenario, TwinScenarios, TwinToolCall, test_context, twin_openai};
use fabro_auth::{AuthCredential, AuthDetails};
use fabro_config::Storage;
use fabro_model::Provider;
use fabro_test::{
TestMode, TwinOpenAi, TwinScenario, TwinScenarios, TwinToolCall, test_context, twin_openai,
};
use fabro_vault::{SecretType, Vault};
use super::{find_run_dir, read_conclusion};
use super::read_conclusion;
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
@ -51,6 +57,73 @@ fn stage_provider() -> &'static str {
}
}
fn toml_path(path: &std::path::Path) -> String {
path.display()
.to_string()
.replace('\\', "\\\\")
.replace('"', "\\\"")
}
fn twin_server_storage_dir(context: &fabro_test::TestContext) -> std::path::PathBuf {
context.temp_dir.join("hook-server-storage")
}
fn settings_with_hook(context: &fabro_test::TestContext, hook: &str) -> String {
if TestMode::from_env().is_twin() {
format!(
r#"[server.storage]
root = "{}"
[server.auth]
methods = ["dev-token"]
{hook}"#,
toml_path(&twin_server_storage_dir(context)),
)
} else {
hook.to_string()
}
}
fn write_hook_settings(context: &fabro_test::TestContext, hook: &str) {
let settings = settings_with_hook(context, hook);
if settings.trim().is_empty() {
return;
}
context.write_home(".fabro/settings.toml", settings);
}
fn seed_openai_vault(storage_dir: &std::path::Path, base_url: &str, api_key: &str) {
let mut vault =
Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load");
vault
.set(
"openai",
&serde_json::to_string(&AuthCredential {
provider: Provider::OpenAi,
details: AuthDetails::ApiKey {
key: api_key.to_string(),
},
})
.expect("OpenAI test credential should serialize"),
SecretType::Credential,
None,
)
.expect("OpenAI credential should store in test vault");
vault
.set("OPENAI_BASE_URL", base_url, SecretType::Environment, None)
.expect("OpenAI base URL should store in test vault");
}
fn configure_twin_server(
context: &mut fabro_test::TestContext,
twin: &TwinOpenAi,
namespace: &str,
) {
seed_openai_vault(&twin_server_storage_dir(context), &twin.base_url, namespace);
context.isolated_server();
}
fn write_workflow(context: &fabro_test::TestContext, name: &str, dot: &str) -> std::path::PathBuf {
context.write_temp(name, dot);
context.temp_dir.join(name)
@ -69,25 +142,28 @@ fn configure_hook_env(cmd: &mut assert_cmd::Command, hook_model: &str) {
cmd.arg("--model").arg(hook_model);
}
fn conclusion_status(context: &fabro_test::TestContext) -> String {
let run_dir = find_run_dir(&context);
read_conclusion(&run_dir)["status"]
.as_str()
.expect("conclusion should include a string status")
.to_string()
async fn conclusion_status(context: &fabro_test::TestContext) -> String {
let run_dir = context.single_run_dir();
tokio::task::spawn_blocking(move || {
read_conclusion(&run_dir)["status"]
.as_str()
.expect("conclusion should include a string status")
.to_string()
})
.await
.expect("conclusion status task should complete")
}
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_prompt_proceed_allows_run() {
let context = test_context!();
context.write_home(
".fabro/settings.toml",
let mut context = test_context!();
write_hook_settings(
&context,
&format!(
r#"
[[hooks]]
[[run.hooks]]
name = "prompt-proceed"
event = "run_start"
type = "prompt"
prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}."
model = "{model}"
"#,
@ -111,6 +187,7 @@ model = "{model}"
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
.load(twin)
.await;
configure_twin_server(&mut context, twin, &namespace);
let mut cmd = context.run_cmd();
configure_hook_env(&mut cmd, stage_model());
twin.configure_command(&mut cmd, &namespace);
@ -123,20 +200,19 @@ model = "{model}"
run_success_output(cmd).await;
}
assert_eq!(conclusion_status(&context), "succeeded");
assert_eq!(conclusion_status(&context).await, "succeeded");
}
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_prompt_block_prevents_run() {
let context = test_context!();
context.write_home(
".fabro/settings.toml",
let mut context = test_context!();
write_hook_settings(
&context,
&format!(
r#"
[[hooks]]
[[run.hooks]]
name = "prompt-block"
event = "run_start"
type = "prompt"
prompt = "Check: is 2+2 equal to 5? If the statement is true, respond {{\"ok\": true}}. If false, respond {{\"ok\": false, \"reason\": \"math check failed\"}}."
model = "{model}"
"#,
@ -163,6 +239,7 @@ model = "{model}"
)
.load(twin)
.await;
configure_twin_server(&mut context, twin, &namespace);
let mut cmd = context.run_cmd();
configure_hook_env(&mut cmd, stage_model());
twin.configure_command(&mut cmd, &namespace);
@ -184,18 +261,18 @@ model = "{model}"
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_agent_proceed_allows_run() {
let context = test_context!();
context.write_home(
".fabro/settings.toml",
let mut context = test_context!();
write_hook_settings(
&context,
&format!(
r#"
[[hooks]]
[[run.hooks]]
name = "agent-proceed"
event = "run_start"
type = "agent"
prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}. Do not use any tools."
model = "{model}"
max_tool_rounds = 1
agent = "enabled"
"#,
model = hook_model()
),
@ -217,6 +294,7 @@ max_tool_rounds = 1
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
.load(twin)
.await;
configure_twin_server(&mut context, twin, &namespace);
let mut cmd = context.run_cmd();
configure_hook_env(&mut cmd, stage_model());
twin.configure_command(&mut cmd, &namespace);
@ -229,25 +307,25 @@ max_tool_rounds = 1
run_success_output(cmd).await;
}
assert_eq!(conclusion_status(&context), "succeeded");
assert_eq!(conclusion_status(&context).await, "succeeded");
}
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_agent_with_tool_use() {
let context = test_context!();
let mut context = test_context!();
let marker = context.temp_dir.join("hook_check.txt");
std::fs::write(&marker, "READY").unwrap();
context.write_home(
".fabro/settings.toml",
write_hook_settings(
&context,
&format!(
r#"
[[hooks]]
[[run.hooks]]
name = "agent-tools"
event = "run_start"
type = "agent"
prompt = "Read the file at {path} using the read_file tool. If it contains 'READY', respond with {{\"ok\": true}}. Otherwise respond with {{\"ok\": false, \"reason\": \"not ready\"}}."
model = "{model}"
max_tool_rounds = 5
agent = "enabled"
"#,
path = marker.display(),
model = hook_model()
@ -274,6 +352,7 @@ max_tool_rounds = 5
.scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#))
.load(twin)
.await;
configure_twin_server(&mut context, twin, &namespace);
let mut cmd = context.run_cmd();
configure_hook_env(&mut cmd, stage_model());
twin.configure_command(&mut cmd, &namespace);
@ -286,12 +365,13 @@ max_tool_rounds = 5
run_success_output(cmd).await;
}
assert_eq!(conclusion_status(&context), "succeeded");
assert_eq!(conclusion_status(&context).await, "succeeded");
}
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn arc_e2e_with_real_llm() {
let context = test_context!();
let mut context = test_context!();
write_hook_settings(&context, "");
let hello = context.temp_dir.join("hello.txt");
let workflow = write_workflow(
&context,
@ -328,6 +408,7 @@ async fn arc_e2e_with_real_llm() {
)
.load(twin)
.await;
configure_twin_server(&mut context, twin, &namespace);
let mut cmd = context.run_cmd();
configure_hook_env(&mut cmd, stage_model());
twin.configure_command(&mut cmd, &namespace);
@ -345,5 +426,5 @@ async fn arc_e2e_with_real_llm() {
"Hello from LLM",
"workflow should create the expected file"
);
assert_eq!(conclusion_status(&context), "succeeded");
assert_eq!(conclusion_status(&context).await, "succeeded");
}

View file

@ -3,6 +3,7 @@
reason = "This test module prefers explicit type paths over extra imports."
)]
mod acp;
mod agent_linear;
mod command_agent_mixed;
mod command_pipeline;

View file

@ -1,11 +1,10 @@
use std::sync::Arc;
use std::time::Duration;
use fabro_graphviz::graph::{AttrValue, Node};
use fabro_llm::provider::Provider;
use fabro_workflow::context::Context;
use fabro_workflow::event::Emitter;
use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult};
use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult, CodergenRunRequest};
use fabro_workflow::handler::llm::cli::AgentCliBackend;
/// Run a real CLI tool via LocalSandbox and verify the full flow.
@ -14,8 +13,7 @@ async fn run_real_cli_test(provider: Provider, model: &str) {
let env: Arc<dyn fabro_agent::Sandbox> = Arc::new(fabro_agent::LocalSandbox::new(
workspace.path().to_path_buf(),
));
let backend = AgentCliBackend::new_from_env(model.to_string(), provider)
.with_poll_interval(Duration::from_millis(10));
let backend = AgentCliBackend::new_from_env(model.to_string(), provider);
let mut node = Node::new("real_cli_test");
node.attrs.insert(
@ -26,16 +24,16 @@ async fn run_real_cli_test(provider: Provider, model: &str) {
let context = Context::new();
let emitter = Arc::new(Emitter::default());
let result = backend
.run(
&node,
"What is 2+2? Reply with just the number.",
&context,
None,
&emitter,
&env,
None,
tokio_util::sync::CancellationToken::new(),
)
.run(CodergenRunRequest {
node: &node,
prompt: "What is 2+2? Reply with just the number.",
context: &context,
thread_id: None,
emitter: &emitter,
sandbox: &env,
tool_hooks: None,
cancel_token: tokio_util::sync::CancellationToken::new(),
})
.await
.unwrap_or_else(|_| panic!("CLI backend ({provider}/{model}) should succeed"));

View file

@ -5,7 +5,7 @@
use std::path::PathBuf;
use fabro_cli::{ManifestBuildInput, build_run_manifest};
use fabro_manifest::{ManifestBuildInput, build_run_manifest};
use fabro_workflow::ManifestPath;
#[test]

View file

@ -794,25 +794,27 @@ impl Client {
Ok(bytes)
}
pub async fn start_run(&self, run_id: &RunId, resume: bool) -> Result<()> {
self.send_api(|client| async move {
client
.start_run()
.id(run_id.to_string())
.body(types::StartRunRequest { resume })
.send()
.await
})
.await?;
Ok(())
pub async fn start_run(&self, run_id: &RunId, resume: bool) -> Result<RunSummary> {
let response = self
.send_api(|client| async move {
client
.start_run()
.id(run_id.to_string())
.body(types::StartRunRequest { resume })
.send()
.await
})
.await?;
convert_type(response.into_inner())
}
pub async fn cancel_run(&self, run_id: &RunId) -> Result<()> {
self.send_api(
|client| async move { client.cancel_run().id(run_id.to_string()).send().await },
)
.await?;
Ok(())
pub async fn cancel_run(&self, run_id: &RunId) -> Result<RunSummary> {
let response = self
.send_api(
|client| async move { client.cancel_run().id(run_id.to_string()).send().await },
)
.await?;
convert_type(response.into_inner())
}
pub async fn interrupt_run(&self, run_id: &RunId) -> Result<()> {
@ -844,20 +846,22 @@ impl Client {
Ok(())
}
pub async fn archive_run(&self, run_id: &RunId) -> Result<()> {
self.send_api(
|client| async move { client.archive_run().id(run_id.to_string()).send().await },
)
.await?;
Ok(())
pub async fn archive_run(&self, run_id: &RunId) -> Result<RunSummary> {
let response = self
.send_api(
|client| async move { client.archive_run().id(run_id.to_string()).send().await },
)
.await?;
convert_type(response.into_inner())
}
pub async fn unarchive_run(&self, run_id: &RunId) -> Result<()> {
self.send_api(|client| async move {
client.unarchive_run().id(run_id.to_string()).send().await
})
.await?;
Ok(())
pub async fn unarchive_run(&self, run_id: &RunId) -> Result<RunSummary> {
let response = self
.send_api(
|client| async move { client.unarchive_run().id(run_id.to_string()).send().await },
)
.await?;
convert_type(response.into_inner())
}
pub async fn rewind_run(
@ -1123,6 +1127,50 @@ impl Client {
Ok(all_events)
}
pub async fn list_run_events_until(
&self,
run_id: &RunId,
since_seq: Option<u32>,
max_events: usize,
) -> Result<Vec<EventEnvelope>> {
if max_events == 0 {
return Ok(Vec::new());
}
let mut next_since_seq = since_seq;
let mut all_events = Vec::new();
while all_events.len() < max_events {
let remaining = max_events - all_events.len();
let response = self
.send_api(|client| async move {
let mut request = client
.list_run_events()
.id(run_id.to_string())
.limit(remaining.min(1000) as u64);
if let Some(seq) = next_since_seq.and_then(non_zero_u64_from_u32) {
request = request.since_seq(seq);
}
request.send().await
})
.await?;
let parsed = response.into_inner();
let page_events = parsed
.data
.into_iter()
.map(convert_type::<_, EventEnvelope>)
.collect::<Result<Vec<EventEnvelope>>>()?;
let next_page_since_seq = page_events.last().map(|event| event.seq.saturating_add(1));
all_events.extend(page_events);
if !parsed.meta.has_more || next_page_since_seq.is_none() {
break;
}
next_since_seq = next_page_since_seq;
}
Ok(all_events)
}
pub async fn attach_run_events(
&self,
run_id: &RunId,

View file

@ -354,6 +354,8 @@ pub(crate) fn resolve_mcp_entry(name: &str, entry: &McpEntryLayer) -> McpServerS
McpServerSettings {
name: name.to_string(),
transport,
current_dir: None,
clear_env: false,
startup_timeout_secs,
tool_timeout_secs,
}

View file

@ -0,0 +1,29 @@
[package]
name = "fabro-manifest"
edition.workspace = true
version.workspace = true
publish = false
license.workspace = true
description = "Fabro run manifest construction"
[lib]
doctest = false
[lints]
workspace = true
[dependencies]
anyhow.workspace = true
fabro-api = { path = "../fabro-api" }
fabro-config = { path = "../fabro-config" }
fabro-github = { path = "../fabro-github" }
fabro-graphviz = { path = "../fabro-graphviz" }
fabro-template = { path = "../fabro-template" }
fabro-types = { path = "../fabro-types" }
fabro-workflow = { path = "../fabro-workflow" }
git2.workspace = true
toml.workspace = true
[dev-dependencies]
tempfile = "3"
temp-env = "0.3"

View file

@ -10,19 +10,21 @@ use anyhow::{Context, Result, anyhow};
use fabro_api::types;
use fabro_config::project::{self, discover_project_config, resolve_workflow_path};
use fabro_config::run::{resolve_run_goal_from_layer, resolve_run_goal_from_namespace};
use fabro_config::{CliLayer, DaytonaDockerfileLayer, RunLayer, WorkflowSettingsBuilder};
use fabro_config::{
CliLayer, DaytonaDockerfileLayer, DockerSandboxLayer, ReplaceMap, RunExecutionLayer,
RunGoalLayer, RunLayer, RunModelLayer, RunSandboxLayer, WorkflowSettingsBuilder,
};
use fabro_graphviz::graph::AttrValue;
use fabro_graphviz::parser;
use fabro_template::{TemplateContext, render as render_template};
use fabro_types::settings::run::{ResolvedGoalSource, ResolvedRunGoal};
use fabro_types::settings::interp::InterpString;
use fabro_types::settings::run::{ApprovalMode, ResolvedGoalSource, ResolvedRunGoal, RunMode};
use fabro_types::{DirtyStatus, GitContext, PreRunPushOutcome, RunId, WorkflowSettings};
use fabro_workflow::ManifestPath;
use fabro_workflow::git::{
GitSyncStatus, branch_needs_push, head_sha, push_branch_noninteractive, sync_status,
};
use crate::args::{PreflightArgs, RunArgs};
#[derive(Debug, Default)]
pub struct ManifestBuildInput {
pub workflow: PathBuf,
@ -43,6 +45,81 @@ pub struct BuiltManifest {
pub target_path: PathBuf,
}
#[derive(Debug, Default)]
pub struct RunOverrideInput<'a> {
pub goal: Option<&'a str>,
pub model: Option<&'a str>,
pub provider: Option<&'a str>,
pub sandbox: Option<&'a str>,
pub docker_image: Option<&'a str>,
pub preserve_sandbox: Option<bool>,
pub dry_run: Option<bool>,
pub auto_approve: Option<bool>,
pub labels: HashMap<String, String>,
}
#[must_use]
pub fn build_run_overrides(input: RunOverrideInput<'_>) -> RunLayer {
let goal = input
.goal
.map(|goal| RunGoalLayer::Inline(InterpString::parse(goal)));
let model = (input.model.is_some() || input.provider.is_some()).then(|| RunModelLayer {
provider: input.provider.map(InterpString::parse),
name: input.model.map(InterpString::parse),
fallbacks: Vec::new(),
controls: None,
});
let sandbox = (input.sandbox.is_some()
|| input.docker_image.is_some()
|| input.preserve_sandbox.is_some())
.then(|| RunSandboxLayer {
provider: input.sandbox.map(ToOwned::to_owned),
docker: input.docker_image.map(|image| DockerSandboxLayer {
image: Some(image.to_string()),
..DockerSandboxLayer::default()
}),
preserve: input.preserve_sandbox,
..RunSandboxLayer::default()
});
let execution =
(input.dry_run.is_some() || input.auto_approve.is_some()).then(|| RunExecutionLayer {
mode: input.dry_run.map(|dry_run| {
if dry_run {
RunMode::DryRun
} else {
RunMode::Normal
}
}),
approval: input.auto_approve.map(|auto_approve| {
if auto_approve {
ApprovalMode::Auto
} else {
ApprovalMode::Prompt
}
}),
});
RunLayer {
goal,
metadata: ReplaceMap::from(input.labels),
model,
sandbox,
execution,
..RunLayer::default()
}
}
#[must_use]
pub fn build_sparse_run_overrides(input: RunOverrideInput<'_>) -> Option<RunLayer> {
let run = build_run_overrides(input);
(run.goal.is_some()
|| !run.metadata.is_empty()
|| run.model.is_some()
|| run.sandbox.is_some()
|| run.execution.is_some())
.then_some(run)
}
struct CollectContext<'a> {
cwd: &'a Path,
inputs: &'a HashMap<String, toml::Value>,
@ -172,42 +249,6 @@ pub fn build_run_manifest(input: ManifestBuildInput) -> Result<BuiltManifest> {
})
}
pub(crate) fn run_manifest_args(args: &RunArgs) -> Option<types::ManifestArgs> {
let payload = types::ManifestArgs {
auto_approve: args.auto_approve.then_some(true),
dry_run: args.dry_run.then_some(true),
label: args.label.clone(),
model: args.model.clone(),
preserve_sandbox: args.preserve_sandbox.then_some(true),
provider: args.provider.clone(),
sandbox: args
.sandbox
.map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()),
docker_image: None,
input: args.inputs.values.clone(),
verbose: args.verbose.then_some(true),
};
(!manifest_args_is_empty(&payload)).then_some(payload)
}
pub(crate) fn preflight_manifest_args(args: &PreflightArgs) -> Option<types::ManifestArgs> {
let payload = types::ManifestArgs {
auto_approve: None,
dry_run: None,
label: Vec::new(),
model: args.model.clone(),
preserve_sandbox: None,
provider: args.provider.clone(),
sandbox: args
.sandbox
.map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()),
docker_image: None,
input: args.inputs.values.clone(),
verbose: args.verbose.then_some(true),
};
(!manifest_args_is_empty(&payload)).then_some(payload)
}
fn collect_workflow_entry(
context: &mut CollectContext<'_>,
workflow: &Path,
@ -650,7 +691,7 @@ fn manifest_path_from_absolute(path: &Path, cwd: &Path) -> Result<ManifestPath>
.ok_or_else(|| anyhow!("Failed to compute manifest path for {}", path.display()))
}
fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool {
pub fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool {
args.auto_approve.is_none()
&& args.dry_run.is_none()
&& args.label.is_empty()
@ -667,6 +708,65 @@ fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool {
mod tests {
use super::*;
#[test]
fn build_run_overrides_sets_common_cli_and_mcp_layers() {
let overrides = build_run_overrides(RunOverrideInput {
goal: Some("ship it"),
model: Some("gpt-5.4-mini"),
provider: Some("openai"),
sandbox: Some("local"),
docker_image: None,
preserve_sandbox: Some(true),
dry_run: Some(true),
auto_approve: Some(false),
labels: [("source".to_string(), "mcp".to_string())]
.into_iter()
.collect(),
});
let goal = overrides.goal.expect("goal override");
assert!(matches!(goal, fabro_config::RunGoalLayer::Inline(_)));
assert_eq!(
overrides
.model
.as_ref()
.unwrap()
.name
.as_ref()
.unwrap()
.as_source(),
"gpt-5.4-mini"
);
assert_eq!(
overrides
.model
.as_ref()
.unwrap()
.provider
.as_ref()
.unwrap()
.as_source(),
"openai"
);
assert_eq!(
overrides.sandbox.as_ref().unwrap().provider.as_deref(),
Some("local")
);
assert_eq!(overrides.sandbox.as_ref().unwrap().preserve, Some(true));
assert_eq!(
overrides.execution.as_ref().unwrap().mode,
Some(RunMode::DryRun)
);
assert_eq!(
overrides.execution.as_ref().unwrap().approval,
Some(ApprovalMode::Prompt)
);
assert_eq!(
overrides.metadata.0.get("source").map(String::as_str),
Some("mcp")
);
}
#[test]
fn build_manifest_bundles_imports_prompts_and_children() {
let temp = tempfile::tempdir().unwrap();

View file

@ -0,0 +1,31 @@
[package]
name = "fabro-mcp-server"
edition.workspace = true
version.workspace = true
publish = false
license.workspace = true
description = "Fabro MCP stdio server"
[lib]
doctest = false
[lints]
workspace = true
[dependencies]
anyhow.workspace = true
chrono = { workspace = true, features = ["serde"] }
fabro-api = { path = "../fabro-api" }
fabro-client = { path = "../fabro-client" }
fabro-manifest = { path = "../fabro-manifest" }
fabro-config = { path = "../fabro-config" }
fabro-server = { path = "../fabro-server" }
fabro-types = { path = "../fabro-types" }
fabro-util = { path = "../fabro-util" }
futures.workspace = true
rmcp = { workspace = true, features = ["server", "macros", "schemars", "transport-io"] }
schemars = "1.2.1"
serde.workspace = true
serde_json.workspace = true
tokio.workspace = true
toml.workspace = true

View file

@ -0,0 +1,146 @@
#![expect(
clippy::disallowed_methods,
reason = "MCP client config setup intentionally performs small synchronous JSON file reads/writes from a CLI command."
)]
use std::path::{Path, PathBuf};
use anyhow::{Context as _, Result, anyhow};
use serde_json::map::Entry;
use serde_json::{Map, Value, json};
use crate::{McpAgent, McpConfigSettings, McpInitSettings};
const SERVER_NAME: &str = "fabro";
pub fn config_json(settings: &McpConfigSettings) -> Result<String> {
serde_json::to_string_pretty(&generic_config(settings))
.map(|json| format!("{json}\n"))
.context("failed to render Fabro MCP client config")
}
pub fn init_agent(settings: &McpInitSettings) -> Result<()> {
let entry = server_entry(&settings.config);
for path in agent_config_paths(settings.agent, &settings.home_dir) {
merge_server_entry(&path, entry.clone())?;
}
Ok(())
}
fn generic_config(settings: &McpConfigSettings) -> Value {
json!({
"mcpServers": {
SERVER_NAME: server_entry(settings)
}
})
}
fn server_entry(settings: &McpConfigSettings) -> Value {
json!({
"command": "fabro",
"args": start_args(settings),
})
}
fn start_args(settings: &McpConfigSettings) -> Vec<String> {
let mut args = vec!["mcp".to_string(), "start".to_string()];
if let Some(server) = settings.server.as_ref() {
args.push("--server".to_string());
args.push(server.clone());
}
if let Some(storage_dir) = settings.storage_dir.as_deref() {
args.push("--storage-dir".to_string());
args.push(storage_dir.display().to_string());
}
args
}
fn merge_server_entry(path: &Path, entry: Value) -> Result<()> {
if let Some(parent) = path.parent() {
std::fs::create_dir_all(parent)
.with_context(|| format!("failed to create {}", parent.display()))?;
}
let mut root = match std::fs::read_to_string(path) {
Ok(contents) => serde_json::from_str::<Value>(&contents)
.with_context(|| format!("failed to parse MCP config {}", path.display()))?,
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Value::Object(Map::new()),
Err(err) => return Err(err).with_context(|| format!("failed to read {}", path.display())),
};
let root_object = root
.as_object_mut()
.ok_or_else(|| anyhow!("MCP config {} must contain a JSON object", path.display()))?;
let servers = match root_object.entry("mcpServers") {
Entry::Vacant(entry) => entry.insert(Value::Object(Map::new())),
Entry::Occupied(entry) => entry.into_mut(),
};
let servers_object = servers.as_object_mut().ok_or_else(|| {
anyhow!(
"MCP config {} field mcpServers must contain a JSON object",
path.display()
)
})?;
servers_object.insert(SERVER_NAME.to_string(), entry);
let rendered = serde_json::to_string_pretty(&root)
.map(|json| format!("{json}\n"))
.with_context(|| format!("failed to render MCP config {}", path.display()))?;
std::fs::write(path, rendered).with_context(|| format!("failed to write {}", path.display()))
}
fn agent_config_paths(agent: McpAgent, home_dir: &Path) -> Vec<PathBuf> {
match agent {
McpAgent::Claude => vec![
claude_desktop_config_path(home_dir),
claude_code_config_path(home_dir),
],
McpAgent::Cursor => vec![home_dir.join(".cursor").join("mcp.json")],
McpAgent::Windsurf => vec![
home_dir
.join(".codeium")
.join("windsurf")
.join("mcp_config.json"),
],
}
}
fn claude_code_config_path(home_dir: &Path) -> PathBuf {
home_dir.join(".claude.json")
}
fn claude_desktop_config_path(home_dir: &Path) -> PathBuf {
#[cfg(target_os = "macos")]
{
home_dir
.join("Library")
.join("Application Support")
.join("Claude")
.join("claude_desktop_config.json")
}
#[cfg(target_os = "linux")]
{
home_dir
.join(".config")
.join("Claude")
.join("claude_desktop_config.json")
}
#[cfg(target_os = "windows")]
{
let app_data = std::env::var_os("APPDATA")
.map(PathBuf::from)
.unwrap_or_else(|| home_dir.join("AppData").join("Roaming"));
app_data.join("Claude").join("claude_desktop_config.json")
}
#[cfg(not(any(target_os = "macos", target_os = "linux", target_os = "windows")))]
{
home_dir
.join(".config")
.join("Claude")
.join("claude_desktop_config.json")
}
}

View file

@ -0,0 +1,55 @@
mod config;
mod run_tools;
mod server;
use std::future::Future;
use std::path::PathBuf;
use std::pin::Pin;
use std::sync::Arc;
use anyhow::Result;
pub use config::{config_json, init_agent};
use fabro_client::Client;
pub use server::start;
pub type FabroClientFuture = Pin<Box<dyn Future<Output = Result<Client>> + Send>>;
pub type FabroClientFactory = Arc<dyn Fn() -> FabroClientFuture + Send + Sync>;
#[derive(Clone)]
pub struct FabroMcpServerSettings {
pub client_factory: FabroClientFactory,
pub config_path: PathBuf,
pub cwd: PathBuf,
}
impl std::fmt::Debug for FabroMcpServerSettings {
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter
.debug_struct("FabroMcpServerSettings")
.field("client_factory", &"<factory>")
.field("config_path", &self.config_path)
.field("cwd", &self.cwd)
.finish()
}
}
#[derive(Debug, Clone, Default)]
pub struct McpConfigSettings {
pub server: Option<String>,
pub storage_dir: Option<PathBuf>,
}
#[derive(Debug, Clone)]
pub struct McpInitSettings {
pub agent: McpAgent,
pub config: McpConfigSettings,
pub home_dir: PathBuf,
}
#[derive(Debug, Clone, Copy)]
pub enum McpAgent {
Claude,
Cursor,
Windsurf,
}

View file

@ -0,0 +1,21 @@
#![allow(
dead_code,
reason = "MCP DTO fields are consumed by serde and schema generation even when not read directly."
)]
mod common;
mod create;
mod events;
mod gather;
mod interact;
mod manifest;
mod search;
pub(crate) use common::{ToolError, error_result, success_result};
pub(crate) use create::{FabroRunCreateParams, ValidatedCreateRuns, create_runs, create_runs_text};
pub(crate) use events::{FabroRunEventsParams, ValidatedRunEvents, run_events, run_events_text};
pub(crate) use gather::{FabroRunGatherParams, ValidatedGatherRuns, gather_runs, gather_runs_text};
pub(crate) use interact::{
FabroRunInteractParams, ValidatedInteractRun, interact_run, interact_run_text,
};
pub(crate) use search::{FabroRunSearchParams, ValidatedSearchRuns, search_runs, search_runs_text};

View file

@ -0,0 +1,141 @@
use std::collections::HashMap;
use chrono::{DateTime, NaiveDate, Utc};
use fabro_client::Client;
use fabro_types::{Run, RunId, RunStatus};
use fabro_util::exit::{self, ExitClass};
use rmcp::model::{CallToolResult, Content};
use schemars::JsonSchema;
use serde::Serialize;
#[derive(Debug)]
pub(crate) struct ToolError {
message: String,
}
impl ToolError {
pub(crate) fn message(message: impl Into<String>) -> Self {
Self {
message: message.into(),
}
}
pub(crate) fn from_anyhow(err: &anyhow::Error) -> Self {
Self::message(format_tool_error(err))
}
pub(crate) fn as_str(&self) -> &str {
&self.message
}
}
pub(super) type ToolResult<T> = Result<T, ToolError>;
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct RunSummaryResult {
pub(crate) run_id: String,
pub(crate) workflow_name: String,
pub(crate) workflow_slug: Option<String>,
pub(crate) status: String,
pub(crate) archived: bool,
pub(crate) created_at: String,
pub(crate) started_at: Option<String>,
pub(crate) completed_at: Option<String>,
pub(crate) labels: HashMap<String, String>,
pub(crate) source_directory: Option<String>,
pub(crate) repo_origin_url: Option<String>,
pub(crate) goal: String,
}
pub(crate) fn success_result<T: Serialize>(
value: &T,
text: impl Into<String>,
) -> Result<CallToolResult, rmcp::ErrorData> {
let structured_content = serde_json::to_value(value).map_err(|err| {
rmcp::ErrorData::internal_error(
format!("failed to serialize Fabro MCP tool result: {err}"),
None,
)
})?;
let mut result = CallToolResult::structured(structured_content);
result.content = vec![Content::text(text.into())];
Ok(result)
}
pub(crate) fn error_result(err: ToolError) -> CallToolResult {
CallToolResult::error(vec![Content::text(err.message)])
}
pub(super) fn validate_len(name: &str, len: usize, min: usize, max: usize) -> ToolResult<()> {
if len < min {
return Err(ToolError::message(format!(
"{name} must contain at least {min} item(s)"
)));
}
if len > max {
return Err(ToolError::message(format!(
"{name} must contain no more than {max} item(s)"
)));
}
Ok(())
}
pub(super) async fn retrieve_run(client: &Client, run_id: &RunId) -> ToolResult<Run> {
client
.retrieve_run(run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))
}
pub(super) fn run_summary_result(run: &Run) -> RunSummaryResult {
RunSummaryResult {
run_id: run.id.to_string(),
workflow_name: run.workflow.name.clone(),
workflow_slug: run.workflow.slug.clone(),
status: run_status_kind(run.lifecycle.status).to_string(),
archived: run.lifecycle.archived,
created_at: run.timestamps.created_at.to_rfc3339(),
started_at: run
.timestamps
.started_at
.map(|timestamp| timestamp.to_rfc3339()),
completed_at: run
.timestamps
.completed_at
.map(|timestamp| timestamp.to_rfc3339()),
labels: run.labels.clone(),
source_directory: run.source_directory.clone(),
repo_origin_url: run
.repository
.as_ref()
.and_then(|repository| repository.origin_url.clone()),
goal: run.goal.clone(),
}
}
pub(super) fn parse_datetime_filter(name: &str, raw: &str) -> ToolResult<DateTime<Utc>> {
if let Ok(timestamp) = DateTime::parse_from_rfc3339(raw) {
return Ok(timestamp.with_timezone(&Utc));
}
let date = NaiveDate::parse_from_str(raw, "%Y-%m-%d").map_err(|err| {
ToolError::message(format!("{name} must be RFC3339 or YYYY-MM-DD: {err}"))
})?;
let datetime = date
.and_hms_opt(0, 0, 0)
.ok_or_else(|| ToolError::message(format!("{name} contains an invalid date")))?;
Ok(DateTime::from_naive_utc_and_offset(datetime, Utc))
}
pub(super) fn run_status_kind(status: RunStatus) -> &'static str {
status.kind().into()
}
fn format_tool_error(err: &anyhow::Error) -> String {
let mut rendered = format!("{err:#}");
if exit::exit_class_for(err) == Some(ExitClass::AuthRequired)
&& !rendered.contains("fabro auth login")
{
rendered.push_str("\nRun `fabro auth login` to authenticate.");
}
rendered
}

View file

@ -0,0 +1,231 @@
use std::borrow::Cow;
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use fabro_client::Client;
use fabro_types::RunId;
use schemars::{JsonSchema, Schema, SchemaGenerator, json_schema};
use serde::{Deserialize, Serialize};
use serde_json::Value;
use super::common::{ToolError, ToolResult};
use super::{common, manifest};
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct FabroRunCreateParams {
pub(crate) runs: Vec<CreateRunSpec>,
}
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct CreateRunSpec {
pub(crate) workflow: String,
pub(crate) cwd: Option<PathBuf>,
pub(crate) run_id: Option<String>,
pub(crate) goal: Option<String>,
#[serde(default)]
pub(crate) inputs: HashMap<String, RunInputValue>,
#[serde(default)]
pub(crate) labels: HashMap<String, String>,
pub(crate) dry_run: Option<bool>,
pub(crate) auto_approve: Option<bool>,
pub(crate) model: Option<String>,
pub(crate) provider: Option<String>,
pub(crate) sandbox: Option<String>,
pub(crate) preserve_sandbox: Option<bool>,
pub(crate) start: Option<bool>,
}
#[derive(Debug, Deserialize)]
#[serde(transparent)]
pub(crate) struct RunInputValue(Value);
impl From<Value> for RunInputValue {
fn from(value: Value) -> Self {
Self(value)
}
}
impl RunInputValue {
fn into_inner(self) -> Value {
self.0
}
}
impl JsonSchema for RunInputValue {
fn inline_schema() -> bool {
true
}
fn schema_name() -> Cow<'static, str> {
"RunInputValue".into()
}
fn json_schema(_: &mut SchemaGenerator) -> Schema {
json_schema!({
"description": "Run input override value. Inputs are TOML-compatible scalar values: string, boolean, integer, or float.",
"anyOf": [
{ "type": "string" },
{ "type": "boolean" },
{ "type": "integer" },
{ "type": "number" }
]
})
}
}
#[derive(Debug)]
pub(crate) struct ValidatedCreateRuns {
pub(crate) runs: Vec<ValidatedCreateRunSpec>,
}
#[derive(Debug)]
pub(crate) struct ValidatedCreateRunSpec {
pub(crate) workflow: String,
pub(crate) cwd: Option<PathBuf>,
pub(crate) run_id: Option<RunId>,
pub(crate) goal: Option<String>,
pub(crate) inputs: HashMap<String, toml::Value>,
pub(crate) labels: HashMap<String, String>,
pub(crate) dry_run: Option<bool>,
pub(crate) auto_approve: Option<bool>,
pub(crate) model: Option<String>,
pub(crate) provider: Option<String>,
pub(crate) sandbox: Option<String>,
pub(crate) preserve_sandbox: Option<bool>,
pub(crate) start: Option<bool>,
}
impl TryFrom<FabroRunCreateParams> for ValidatedCreateRuns {
type Error = ToolError;
fn try_from(params: FabroRunCreateParams) -> Result<Self, Self::Error> {
common::validate_len("runs", params.runs.len(), 1, 50)?;
let runs = params
.runs
.into_iter()
.map(ValidatedCreateRunSpec::try_from)
.collect::<Result<Vec<_>, _>>()?;
Ok(Self { runs })
}
}
impl TryFrom<CreateRunSpec> for ValidatedCreateRunSpec {
type Error = ToolError;
fn try_from(spec: CreateRunSpec) -> Result<Self, Self::Error> {
let run_id = spec
.run_id
.as_deref()
.map(str::parse::<RunId>)
.transpose()
.map_err(|err| {
ToolError::message(format!("run_id must be a valid Fabro run id: {err}"))
})?;
let inputs = spec
.inputs
.into_iter()
.map(|(key, value)| {
let value = value.into_inner();
manifest::json_to_toml_value(&key, &value).map(|value| (key, value))
})
.collect::<ToolResult<HashMap<_, _>>>()?;
Ok(Self {
workflow: spec.workflow,
cwd: spec.cwd,
run_id,
goal: spec.goal,
inputs,
labels: spec.labels,
dry_run: spec.dry_run,
auto_approve: spec.auto_approve,
model: spec.model,
provider: spec.provider,
sandbox: spec.sandbox,
preserve_sandbox: spec.preserve_sandbox,
start: spec.start,
})
}
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct CreateRunsResult {
pub(crate) runs: Vec<CreatedRunResult>,
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct CreatedRunResult {
pub(crate) run_id: String,
pub(crate) workflow: String,
pub(crate) started: bool,
pub(crate) status: String,
}
pub(crate) async fn create_runs(
client: Arc<Client>,
base_cwd: &Path,
user_settings_path: &Path,
params: ValidatedCreateRuns,
) -> ToolResult<CreateRunsResult> {
let mut created = Vec::with_capacity(params.runs.len());
for spec in params.runs {
let cwd = spec.cwd.clone().unwrap_or_else(|| base_cwd.to_path_buf());
let manifest = manifest::build_mcp_run_manifest(&spec, &cwd, user_settings_path)?;
let run_id = client
.create_run_from_manifest(manifest)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
let started = spec.start.unwrap_or(true);
let summary = if started {
client
.start_run(&run_id, false)
.await
.map_err(|err| ToolError::from_anyhow(&err))?
} else {
client
.retrieve_run(&run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?
};
created.push(CreatedRunResult {
run_id: summary.id.to_string(),
workflow: spec.workflow,
started,
status: common::run_status_kind(summary.lifecycle.status).to_string(),
});
}
Ok(CreateRunsResult { runs: created })
}
pub(crate) fn create_runs_text(result: &CreateRunsResult) -> String {
let started = result.runs.iter().filter(|run| run.started).count();
format!(
"created {} Fabro run(s), started {started}",
result.runs.len()
)
}
#[cfg(test)]
mod tests {
use schemars::SchemaGenerator;
use serde_json::json;
use super::*;
#[test]
fn run_input_value_schema_allows_only_json_scalars() {
let mut generator = SchemaGenerator::default();
let schema = RunInputValue::json_schema(&mut generator);
let schema = serde_json::to_value(schema).expect("schema should serialize");
assert_eq!(
schema["anyOf"],
json!([
{ "type": "string" },
{ "type": "boolean" },
{ "type": "integer" },
{ "type": "number" },
])
);
}
}

View file

@ -0,0 +1,313 @@
use std::sync::Arc;
use chrono::{DateTime, Utc};
use fabro_client::Client;
use fabro_types::EventEnvelope;
use schemars::JsonSchema;
use serde::{Deserialize, Serialize};
use serde_json::Value;
use super::common;
use super::common::{ToolError, ToolResult};
#[derive(Debug, Clone, Copy, Deserialize, Serialize, JsonSchema)]
#[serde(rename_all = "snake_case")]
pub(crate) enum RunEventsAction {
List,
Details,
Search,
}
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct FabroRunEventsParams {
pub(crate) action: RunEventsAction,
pub(crate) run_id: String,
pub(crate) event_types: Option<Vec<String>>,
pub(crate) categories: Option<Vec<String>>,
pub(crate) direction: Option<String>,
pub(crate) created_after: Option<String>,
pub(crate) created_before: Option<String>,
pub(crate) first: Option<usize>,
pub(crate) after: Option<u32>,
pub(crate) event_ids: Option<Vec<String>>,
pub(crate) offset: Option<usize>,
pub(crate) limit: Option<usize>,
pub(crate) max_content_length: Option<usize>,
pub(crate) query: Option<String>,
}
#[derive(Debug)]
pub(crate) struct ValidatedRunEvents {
pub(crate) raw: FabroRunEventsParams,
pub(crate) descending: bool,
pub(crate) first: usize,
pub(crate) created_after: Option<DateTime<Utc>>,
pub(crate) created_before: Option<DateTime<Utc>>,
}
impl TryFrom<FabroRunEventsParams> for ValidatedRunEvents {
type Error = ToolError;
fn try_from(params: FabroRunEventsParams) -> Result<Self, Self::Error> {
if params.run_id.trim().is_empty() {
return Err(ToolError::message("run_id is required"));
}
let first = params.first.or(params.limit).unwrap_or(50);
if first > 200 {
return Err(ToolError::message("first must be <= 200"));
}
let descending = match params.direction.as_deref() {
None | Some("asc") => false,
Some("desc") => true,
Some(_) => return Err(ToolError::message("direction must be `asc` or `desc`")),
};
let created_after = params
.created_after
.as_deref()
.map(|created_after| common::parse_datetime_filter("created_after", created_after))
.transpose()?;
let created_before = params
.created_before
.as_deref()
.map(|created_before| common::parse_datetime_filter("created_before", created_before))
.transpose()?;
if matches!(params.action, RunEventsAction::Details)
&& params.event_ids.as_ref().is_none_or(Vec::is_empty)
{
return Err(ToolError::message(
"event_ids is required for details action",
));
}
if matches!(params.action, RunEventsAction::Search)
&& params
.query
.as_deref()
.is_none_or(|query| query.trim().is_empty())
{
return Err(ToolError::message("query is required for search action"));
}
Ok(Self {
raw: params,
descending,
first,
created_after,
created_before,
})
}
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct RunEventsResult {
pub(crate) run_id: String,
pub(crate) action: RunEventsAction,
pub(crate) events: Vec<RunEventResult>,
pub(crate) next_cursor: Option<u32>,
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct RunEventResult {
pub(crate) event_id: String,
pub(crate) sequence: u32,
pub(crate) event: Value,
pub(crate) truncated: bool,
}
pub(crate) async fn run_events(
client: Arc<Client>,
params: ValidatedRunEvents,
) -> ToolResult<RunEventsResult> {
let descending = params.descending;
let first = params.first;
let created_after = params.created_after;
let created_before = params.created_before;
let raw = params.raw;
let run_id = client
.resolve_run(&raw.run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?
.id;
let fetch_after = if descending { None } else { raw.after };
let mut events = if let Some(limit) = event_fetch_limit(&raw, first) {
client
.list_run_events_until(&run_id, fetch_after, limit)
.await
} else {
client.list_run_events(&run_id, fetch_after, None).await
}
.map_err(|err| ToolError::from_anyhow(&err))?;
if descending {
if let Some(after) = raw.after {
events.retain(|event| event.seq < after);
}
}
filter_events(&mut events, &raw, created_after, created_before);
if descending {
events.reverse();
}
let offset = raw.offset.unwrap_or(0);
let page = events
.into_iter()
.skip(offset)
.take(first)
.collect::<Vec<_>>();
let max_content_length = raw.max_content_length.unwrap_or(20_000);
let results = page
.iter()
.map(|event| run_event_result(event, max_content_length))
.collect::<ToolResult<Vec<_>>>()?;
let next_cursor = page.last().map(|event| {
if descending {
event.seq
} else {
event.seq.saturating_add(1)
}
});
Ok(RunEventsResult {
run_id: run_id.to_string(),
action: raw.action,
events: results,
next_cursor,
})
}
pub(crate) fn run_events_text(result: &RunEventsResult) -> String {
format!("returned {} Fabro event(s)", result.events.len())
}
fn event_fetch_limit(params: &FabroRunEventsParams, first: usize) -> Option<usize> {
let needs_full_scan = params.event_ids.is_some()
|| params.event_types.is_some()
|| params.categories.is_some()
|| params.created_after.is_some()
|| params.created_before.is_some()
|| params.direction.as_deref() == Some("desc")
|| matches!(
params.action,
RunEventsAction::Details | RunEventsAction::Search
);
if needs_full_scan {
return None;
}
let requested = first.saturating_add(params.offset.unwrap_or(0));
Some(requested.max(1))
}
fn filter_events(
events: &mut Vec<EventEnvelope>,
params: &FabroRunEventsParams,
created_after: Option<DateTime<Utc>>,
created_before: Option<DateTime<Utc>>,
) {
if let Some(event_ids) = params.event_ids.as_ref() {
events.retain(|event| event_ids.contains(&event.event.id));
}
if let Some(event_types) = params.event_types.as_ref() {
events.retain(|event| {
event_types
.iter()
.any(|event_type| event_type == event.event.event_name())
});
}
if let Some(categories) = params.categories.as_ref() {
events.retain(|event| {
let category = event
.event
.event_name()
.split('.')
.next()
.unwrap_or_default();
categories.iter().any(|candidate| candidate == category)
});
}
if let Some(cutoff) = created_after {
events.retain(|event| event.event.ts >= cutoff);
}
if let Some(cutoff) = created_before {
events.retain(|event| event.event.ts <= cutoff);
}
if matches!(params.action, RunEventsAction::Search) {
if let Some(query) = params.query.as_deref() {
events.retain(|event| {
serde_json::to_string(event).is_ok_and(|serialized| serialized.contains(query))
});
}
}
}
fn run_event_result(
event: &EventEnvelope,
max_content_length: usize,
) -> ToolResult<RunEventResult> {
let mut serialized = serde_json::to_string(event)
.map_err(|err| ToolError::message(format!("failed to serialize event: {err}")))?;
let truncated = serialized.len() > max_content_length;
let event_value = if truncated {
serialized.truncate(floor_char_boundary(&serialized, max_content_length));
Value::String(serialized)
} else {
serde_json::to_value(event)
.map_err(|err| ToolError::message(format!("failed to serialize event: {err}")))?
};
Ok(RunEventResult {
event_id: event.event.id.clone(),
sequence: event.seq,
event: event_value,
truncated,
})
}
fn floor_char_boundary(value: &str, max_len: usize) -> usize {
let mut boundary = max_len.min(value.len());
while !value.is_char_boundary(boundary) {
boundary -= 1;
}
boundary
}
#[cfg(test)]
mod tests {
use chrono::Utc;
use fabro_types::{EventBody, EventEnvelope, RunEvent, fixtures};
use serde_json::{Value, json};
use super::*;
#[test]
fn run_event_result_truncates_at_utf8_boundary() {
let event = EventEnvelope {
seq: 1,
event: RunEvent {
id: "evt_utf8".to_string(),
ts: Utc::now(),
run_id: fixtures::RUN_1,
node_id: None,
node_label: None,
stage_id: None,
parallel_group_id: None,
parallel_branch_id: None,
session_id: None,
parent_session_id: None,
tool_call_id: None,
actor: None,
body: EventBody::Unknown {
name: "test.utf8".to_string(),
properties: json!({ "message": "éééé" }),
},
},
};
let serialized = serde_json::to_string(&event).unwrap();
let first_multibyte = serialized
.find('é')
.expect("serialized event should contain é");
let result = run_event_result(&event, first_multibyte + 1).unwrap();
assert!(result.truncated);
let Value::String(event_json) = result.event else {
panic!("truncated events should return string payloads");
};
assert!(event_json.is_char_boundary(event_json.len()));
}
}

View file

@ -0,0 +1,109 @@
use std::sync::Arc;
use std::time::{Duration, Instant};
use fabro_client::Client;
use futures::future::try_join_all;
use schemars::JsonSchema;
use serde::{Deserialize, Serialize};
use tokio::time;
use super::common;
use super::common::{RunSummaryResult, ToolError, ToolResult};
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct FabroRunGatherParams {
pub(crate) run_ids: Vec<String>,
pub(crate) timeout_seconds: Option<u64>,
pub(crate) poll_interval_seconds: Option<u64>,
}
#[derive(Debug)]
pub(crate) struct ValidatedGatherRuns {
pub(crate) run_ids: Vec<String>,
pub(crate) timeout_seconds: u64,
pub(crate) poll_interval_seconds: u64,
}
impl TryFrom<FabroRunGatherParams> for ValidatedGatherRuns {
type Error = ToolError;
fn try_from(params: FabroRunGatherParams) -> Result<Self, Self::Error> {
common::validate_len("run_ids", params.run_ids.len(), 1, 50)?;
if params.timeout_seconds.is_some_and(|timeout| timeout > 600) {
return Err(ToolError::message("timeout_seconds must be <= 600"));
}
if params
.poll_interval_seconds
.is_some_and(|interval| interval < 5)
{
return Err(ToolError::message("poll_interval_seconds must be >= 5"));
}
Ok(Self {
run_ids: params.run_ids,
timeout_seconds: params.timeout_seconds.unwrap_or(300),
poll_interval_seconds: params.poll_interval_seconds.unwrap_or(15),
})
}
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct GatherRunsResult {
pub(crate) runs: Vec<RunSummaryResult>,
pub(crate) timed_out: bool,
pub(crate) elapsed_seconds: u64,
}
pub(crate) async fn gather_runs(
client: Arc<Client>,
params: ValidatedGatherRuns,
) -> ToolResult<GatherRunsResult> {
let start = Instant::now();
let deadline = start + Duration::from_secs(params.timeout_seconds);
let run_ids = try_join_all(params.run_ids.into_iter().map(|selector| {
let client = Arc::clone(&client);
async move {
client
.resolve_run(&selector)
.await
.map(|run| run.id)
.map_err(|err| ToolError::from_anyhow(&err))
}
}))
.await?;
loop {
let summaries = try_join_all(run_ids.iter().map(|run_id| {
let client = Arc::clone(&client);
async move { common::retrieve_run(&client, run_id).await }
}))
.await?;
if summaries
.iter()
.all(|run| run.lifecycle.status.is_terminal())
{
return Ok(GatherRunsResult {
runs: summaries.iter().map(common::run_summary_result).collect(),
timed_out: false,
elapsed_seconds: start.elapsed().as_secs(),
});
}
let now = Instant::now();
if now >= deadline {
return Ok(GatherRunsResult {
runs: summaries.iter().map(common::run_summary_result).collect(),
timed_out: true,
elapsed_seconds: start.elapsed().as_secs(),
});
}
let sleep_for = Duration::from_secs(params.poll_interval_seconds).min(deadline - now);
time::sleep(sleep_for).await;
}
}
pub(crate) fn gather_runs_text(result: &GatherRunsResult) -> String {
format!(
"gathered {} Fabro run(s), timed_out={}",
result.runs.len(),
result.timed_out
)
}

View file

@ -0,0 +1,399 @@
use std::borrow::Cow;
use std::sync::Arc;
use fabro_api::types;
use fabro_client::Client;
use fabro_types::RunId;
use schemars::{JsonSchema, Schema, SchemaGenerator, json_schema};
use serde::{Deserialize, Serialize};
use serde_json::{Value, json};
use super::common;
use super::common::{ToolError, ToolResult};
#[derive(Debug, Clone, Copy, Deserialize, Serialize, JsonSchema)]
#[serde(rename_all = "snake_case")]
pub(crate) enum RunInteractAction {
Get,
Start,
Message,
Cancel,
Archive,
Unarchive,
GetQuestions,
Answer,
}
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct FabroRunInteractParams {
pub(crate) action: RunInteractAction,
pub(crate) run_id: String,
pub(crate) message: Option<String>,
pub(crate) interrupt: Option<bool>,
pub(crate) question_id: Option<String>,
pub(crate) answer: Option<AnswerValue>,
}
#[derive(Debug, Deserialize)]
#[serde(transparent)]
pub(crate) struct AnswerValue(Value);
impl From<Value> for AnswerValue {
fn from(value: Value) -> Self {
Self(value)
}
}
impl AnswerValue {
fn into_inner(self) -> Value {
self.0
}
}
impl JsonSchema for AnswerValue {
fn inline_schema() -> bool {
true
}
fn schema_name() -> Cow<'static, str> {
"AnswerValue".into()
}
fn json_schema(_: &mut SchemaGenerator) -> Schema {
json_schema!({
"description": "Answer payload for a pending Fabro question. Use a boolean for yes/no, a string or {\"text\": \"...\"} for freeform text, {\"option\": \"key\"} for a single choice, or {\"options\": [\"key\"]} for multi-select.",
"anyOf": [
{ "type": "boolean" },
{ "type": "string" },
{
"type": "object",
"properties": {
"option": { "type": "string" }
},
"required": ["option"],
"additionalProperties": false
},
{
"type": "object",
"properties": {
"options": {
"type": "array",
"items": { "type": "string" }
}
},
"required": ["options"],
"additionalProperties": false
},
{
"type": "object",
"properties": {
"text": { "type": "string" }
},
"required": ["text"],
"additionalProperties": false
}
]
})
}
}
#[derive(Debug)]
pub(crate) struct ValidatedInteractRun {
pub(crate) run_id: String,
pub(crate) action: ValidatedInteractAction,
}
#[derive(Debug)]
pub(crate) enum ValidatedInteractAction {
Get,
Start,
Message {
message: String,
interrupt: bool,
},
Cancel,
Archive,
Unarchive,
GetQuestions,
Answer {
question_id: String,
body: types::SubmitAnswerRequest,
},
}
impl ValidatedInteractAction {
fn action(&self) -> RunInteractAction {
match self {
Self::Get => RunInteractAction::Get,
Self::Start => RunInteractAction::Start,
Self::Message { .. } => RunInteractAction::Message,
Self::Cancel => RunInteractAction::Cancel,
Self::Archive => RunInteractAction::Archive,
Self::Unarchive => RunInteractAction::Unarchive,
Self::GetQuestions => RunInteractAction::GetQuestions,
Self::Answer { .. } => RunInteractAction::Answer,
}
}
}
impl TryFrom<FabroRunInteractParams> for ValidatedInteractRun {
type Error = ToolError;
fn try_from(params: FabroRunInteractParams) -> Result<Self, Self::Error> {
if params.run_id.trim().is_empty() {
return Err(ToolError::message("run_id is required"));
}
let action = match params.action {
RunInteractAction::Get => ValidatedInteractAction::Get,
RunInteractAction::Start => ValidatedInteractAction::Start,
RunInteractAction::Message => {
let Some(message) = params
.message
.as_deref()
.map(str::trim)
.filter(|message| !message.is_empty())
else {
return Err(ToolError::message("message is required for action message"));
};
ValidatedInteractAction::Message {
message: message.to_string(),
interrupt: params.interrupt.unwrap_or(false),
}
}
RunInteractAction::Cancel => ValidatedInteractAction::Cancel,
RunInteractAction::Archive => ValidatedInteractAction::Archive,
RunInteractAction::Unarchive => ValidatedInteractAction::Unarchive,
RunInteractAction::GetQuestions => ValidatedInteractAction::GetQuestions,
RunInteractAction::Answer => {
let Some(question_id) = params
.question_id
.as_deref()
.map(str::trim)
.filter(|question_id| !question_id.is_empty())
else {
return Err(ToolError::message(
"question_id is required for action answer",
));
};
let Some(answer) = params.answer else {
return Err(ToolError::message("answer is required for action answer"));
};
ValidatedInteractAction::Answer {
question_id: question_id.to_string(),
body: answer_to_submit_request(answer.into_inner())?,
}
}
};
Ok(Self {
run_id: params.run_id.trim().to_string(),
action,
})
}
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct InteractRunResult {
pub(crate) run_id: String,
pub(crate) action: RunInteractAction,
pub(crate) result: Value,
}
pub(crate) async fn interact_run(
client: Arc<Client>,
params: ValidatedInteractRun,
) -> ToolResult<InteractRunResult> {
let run_id = client
.resolve_run(&params.run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?
.id;
let action = params.action.action();
let result = match params.action {
ValidatedInteractAction::Get => interact_get(&client, &run_id).await?,
ValidatedInteractAction::Start => {
let summary = client
.start_run(&run_id, false)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "summary": common::run_summary_result(&summary) })
}
ValidatedInteractAction::Message { message, interrupt } => {
client
.steer_run(&run_id, message.clone(), interrupt)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "message": message, "interrupt": interrupt })
}
ValidatedInteractAction::Cancel => {
let summary = client
.cancel_run(&run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "summary": common::run_summary_result(&summary) })
}
ValidatedInteractAction::Archive => {
let summary = client
.archive_run(&run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "summary": common::run_summary_result(&summary) })
}
ValidatedInteractAction::Unarchive => {
let summary = client
.unarchive_run(&run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "summary": common::run_summary_result(&summary) })
}
ValidatedInteractAction::GetQuestions => {
let questions = client
.list_run_questions(&run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "questions": questions })
}
ValidatedInteractAction::Answer { question_id, body } => {
client
.submit_run_answer(&run_id, &question_id, body)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
json!({ "question_id": question_id, "submitted": true })
}
};
Ok(InteractRunResult {
run_id: run_id.to_string(),
action,
result,
})
}
pub(crate) fn interact_run_text(result: &InteractRunResult) -> String {
format!(
"completed {:?} for Fabro run {}",
result.action, result.run_id
)
}
async fn interact_get(client: &Client, run_id: &RunId) -> ToolResult<Value> {
let summary = common::retrieve_run(client, run_id).await?;
let projection = client
.get_run_state(run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))?;
Ok(json!({
"summary": common::run_summary_result(&summary),
"projection": projection,
}))
}
fn answer_to_submit_request(answer: Value) -> ToolResult<types::SubmitAnswerRequest> {
match answer {
Value::Bool(true) => Ok(types::SubmitAnswerYesRequest {
kind: types::SubmitAnswerYesRequestKind::Yes,
}
.into()),
Value::Bool(false) => Ok(types::SubmitAnswerNoRequest {
kind: types::SubmitAnswerNoRequestKind::No,
}
.into()),
Value::String(text) => Ok(text_answer_request(text)),
Value::Object(mut object) => {
if let Some(option) = object.remove("option") {
let option_key = serde_json::from_value::<String>(option).map_err(|err| {
ToolError::message(format!("answer option must be a string: {err}"))
})?;
Ok(types::SubmitAnswerSelectedRequest {
kind: types::SubmitAnswerSelectedRequestKind::Selected,
option_key,
}
.into())
} else if let Some(options) = object.remove("options") {
let option_keys =
serde_json::from_value::<Vec<String>>(options).map_err(|err| {
ToolError::message(format!("answer options must be strings: {err}"))
})?;
Ok(types::SubmitAnswerMultiSelectedRequest {
kind: types::SubmitAnswerMultiSelectedRequestKind::MultiSelected,
option_keys,
}
.into())
} else if let Some(text) = object.remove("text") {
let text = serde_json::from_value::<String>(text).map_err(|err| {
ToolError::message(format!("answer text must be a string: {err}"))
})?;
Ok(text_answer_request(text))
} else {
Err(ToolError::message(
"answer object must contain one of: option, options, text",
))
}
}
other => Err(ToolError::message(format!(
"unsupported answer value: {other}; expected boolean, string, or object",
))),
}
}
fn text_answer_request(text: String) -> types::SubmitAnswerRequest {
types::SubmitAnswerTextRequest {
kind: types::SubmitAnswerTextRequestKind::Text,
text,
}
.into()
}
#[cfg(test)]
mod tests {
use serde_json::json;
use super::*;
#[test]
fn answer_payloads_map_to_submit_answer_wire_json() {
let cases = [
(json!(true), json!({ "kind": "yes" })),
(json!(false), json!({ "kind": "no" })),
(json!("hello"), json!({ "kind": "text", "text": "hello" })),
(
json!({ "option": "a" }),
json!({ "kind": "selected", "option_key": "a" }),
),
(
json!({ "options": ["a", "b"] }),
json!({ "kind": "multi_selected", "option_keys": ["a", "b"] }),
),
(
json!({ "text": "hello" }),
json!({ "kind": "text", "text": "hello" }),
),
];
for (answer, expected) in cases {
let request = answer_to_submit_request(answer).unwrap();
assert_eq!(serde_json::to_value(request).unwrap(), expected);
}
}
#[test]
fn unsupported_answer_object_is_rejected() {
let err = answer_to_submit_request(json!({ "value": "yes" })).unwrap_err();
assert!(err.as_str().contains("option, options, text"));
}
#[test]
fn interact_answer_validation_rejects_unsupported_json_before_api_calls() {
let err = ValidatedInteractRun::try_from(FabroRunInteractParams {
action: RunInteractAction::Answer,
run_id: "run_123".to_string(),
message: None,
interrupt: None,
question_id: Some("question-1".to_string()),
answer: Some(json!({ "value": "yes" }).into()),
})
.unwrap_err();
assert!(err.as_str().contains("option, options, text"));
}
}

View file

@ -0,0 +1,178 @@
use std::path::{Path, PathBuf};
use fabro_api::types;
use fabro_config::{CliLayer, RunLayer};
use fabro_manifest::{self, ManifestBuildInput, RunOverrideInput};
use fabro_server::manifest_validation;
use serde_json::Value;
use super::common::{ToolError, ToolResult};
use super::create::ValidatedCreateRunSpec;
pub(super) fn build_mcp_run_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult<types::RunManifest> {
let built = fabro_manifest::build_run_manifest(ManifestBuildInput {
workflow: PathBuf::from(&spec.workflow),
cwd: cwd.to_path_buf(),
run_overrides: mcp_run_overrides(spec),
cli_overrides: Some(CliLayer::default()),
input_overrides: spec.inputs.clone(),
args: mcp_manifest_args(spec),
run_id: spec.run_id,
user_settings_path: Some(user_settings_path.to_path_buf()),
})
.map_err(|err| ToolError::from_anyhow(&err))?;
let validation = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)
.map_err(|err| ToolError::from_anyhow(&err))?;
if !validation.ok {
return Err(ToolError::message("workflow manifest validation failed"));
}
Ok(built.manifest)
}
pub(super) fn json_to_toml_value(key: &str, value: &Value) -> ToolResult<toml::Value> {
match value {
Value::Null => Err(ToolError::message(format!(
"input `{key}` cannot be null; use a string, boolean, or number"
))),
Value::Bool(value) => Ok(toml::Value::Boolean(*value)),
Value::Number(value) => {
if let Some(integer) = value.as_i64() {
Ok(toml::Value::Integer(integer))
} else if let Some(float) = value.as_f64() {
Ok(toml::Value::Float(float))
} else {
Err(ToolError::message(format!(
"input `{key}` contains a number outside TOML's supported range"
)))
}
}
Value::String(value) => Ok(toml::Value::String(value.clone())),
Value::Array(_) => Err(ToolError::message(format!(
"input `{key}` does not support array values; use a string, boolean, or number",
))),
Value::Object(_) => Err(ToolError::message(format!(
"input `{key}` does not support object values; use a string, boolean, or number",
))),
}
}
fn mcp_manifest_args(spec: &ValidatedCreateRunSpec) -> Option<types::ManifestArgs> {
let mut input = spec
.inputs
.iter()
.map(|(key, value)| format!("{key}={value}"))
.collect::<Vec<_>>();
input.sort();
let mut label = spec
.labels
.iter()
.map(|(key, value)| format!("{key}={value}"))
.collect::<Vec<_>>();
label.sort();
let payload = types::ManifestArgs {
auto_approve: spec.auto_approve.filter(|value| *value),
docker_image: None,
dry_run: spec.dry_run.filter(|value| *value),
input,
label,
model: spec.model.clone(),
preserve_sandbox: spec.preserve_sandbox.filter(|value| *value),
provider: spec.provider.clone(),
sandbox: spec.sandbox.clone(),
verbose: None,
};
(!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload)
}
fn mcp_run_overrides(spec: &ValidatedCreateRunSpec) -> Option<RunLayer> {
fabro_manifest::build_sparse_run_overrides(RunOverrideInput {
goal: spec.goal.as_deref(),
model: spec.model.as_deref(),
provider: spec.provider.as_deref(),
sandbox: spec.sandbox.as_deref(),
docker_image: None,
preserve_sandbox: spec.preserve_sandbox,
dry_run: spec.dry_run,
auto_approve: spec.auto_approve,
labels: spec.labels.clone(),
})
}
#[cfg(test)]
mod tests {
use std::collections::HashMap;
use serde_json::{Value, json};
use super::super::create::CreateRunSpec;
use super::*;
#[test]
fn json_inputs_convert_scalar_values_to_toml_values() {
let cases = [
(json!("hello"), toml::Value::String("hello".to_string())),
(json!(true), toml::Value::Boolean(true)),
(json!(42), toml::Value::Integer(42)),
(json!(0.5), toml::Value::Float(0.5)),
];
for (json, expected) in cases {
assert_eq!(json_to_toml_value("input", &json).unwrap(), expected);
}
}
#[test]
fn json_input_arrays_and_objects_are_rejected() {
let array_err = json_to_toml_value("matrix", &json!(["a", 1])).unwrap_err();
assert_eq!(
array_err.as_str(),
"input `matrix` does not support array values; use a string, boolean, or number",
);
let object_err = json_to_toml_value("settings", &json!({ "enabled": true })).unwrap_err();
assert_eq!(
object_err.as_str(),
"input `settings` does not support object values; use a string, boolean, or number",
);
}
#[test]
fn json_input_null_is_rejected_with_key_name() {
let err = json_to_toml_value("goal", &Value::Null).unwrap_err();
assert_eq!(
err.as_str(),
"input `goal` cannot be null; use a string, boolean, or number",
);
}
#[test]
fn mcp_manifest_args_preserve_input_provenance() {
let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec {
workflow: "simple".to_string(),
run_id: None,
cwd: None,
goal: None,
inputs: HashMap::from([
("count".to_string(), json!(3).into()),
("decision".to_string(), json!("approve").into()),
]),
labels: HashMap::new(),
model: None,
provider: None,
sandbox: None,
dry_run: None,
auto_approve: None,
preserve_sandbox: None,
start: None,
})
.expect("create spec should validate");
let args = mcp_manifest_args(&spec).expect("input args should be present");
assert_eq!(args.input, vec![r"count=3", r#"decision="approve""#]);
}
}

View file

@ -0,0 +1,378 @@
use std::collections::HashMap;
use std::sync::Arc;
use fabro_client::Client;
use fabro_types::{Run, RunStatusKind};
use futures::future::try_join_all;
use schemars::JsonSchema;
use serde::{Deserialize, Serialize};
use super::common;
use super::common::{RunSummaryResult, ToolError, ToolResult};
const SEARCH_GOAL_PREVIEW_CHARS: usize = 240;
#[derive(Debug, Deserialize, JsonSchema)]
pub(crate) struct FabroRunSearchParams {
pub(crate) run_ids: Option<Vec<String>>,
pub(crate) workflow: Option<String>,
pub(crate) labels: Option<HashMap<String, String>>,
pub(crate) status: Option<Vec<String>>,
pub(crate) archived: Option<bool>,
pub(crate) created_after: Option<String>,
pub(crate) created_before: Option<String>,
pub(crate) first: Option<usize>,
pub(crate) after: Option<String>,
}
#[derive(Debug)]
pub(crate) struct ValidatedSearchRuns {
pub(crate) raw: FabroRunSearchParams,
pub(crate) status: Option<Vec<RunStatusKind>>,
}
impl TryFrom<FabroRunSearchParams> for ValidatedSearchRuns {
type Error = ToolError;
fn try_from(params: FabroRunSearchParams) -> Result<Self, Self::Error> {
if params.first.is_some_and(|first| first > 100) {
return Err(ToolError::message("first must be <= 100"));
}
if let Some(run_ids) = params.run_ids.as_ref() {
common::validate_len("run_ids", run_ids.len(), 1, 100)?;
}
let status = params
.status
.as_ref()
.map(|statuses| {
statuses
.iter()
.map(|status| {
status.parse::<RunStatusKind>().map_err(|_| {
ToolError::message(format!("unknown run status `{status}`"))
})
})
.collect::<ToolResult<Vec<_>>>()
})
.transpose()?;
if let Some(created_after) = params.created_after.as_deref() {
common::parse_datetime_filter("created_after", created_after)?;
}
if let Some(created_before) = params.created_before.as_deref() {
common::parse_datetime_filter("created_before", created_before)?;
}
Ok(Self {
raw: params,
status,
})
}
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct SearchRunsResult {
pub(crate) runs: Vec<SearchRunSummaryResult>,
pub(crate) next_cursor: Option<String>,
}
#[derive(Debug, Serialize, JsonSchema)]
pub(crate) struct SearchRunSummaryResult {
pub(crate) run_id: String,
pub(crate) workflow_name: String,
pub(crate) workflow_slug: Option<String>,
pub(crate) status: String,
pub(crate) archived: bool,
pub(crate) created_at: String,
pub(crate) started_at: Option<String>,
pub(crate) completed_at: Option<String>,
pub(crate) labels: HashMap<String, String>,
pub(crate) source_directory: Option<String>,
pub(crate) repo_origin_url: Option<String>,
pub(crate) goal_preview: String,
pub(crate) goal_truncated: bool,
}
pub(crate) async fn search_runs(
client: Arc<Client>,
params: ValidatedSearchRuns,
) -> ToolResult<SearchRunsResult> {
let status = params.status;
let raw = params.raw;
let runs = if let Some(run_ids) = raw.run_ids.as_ref() {
resolve_requested_runs(&client, run_ids).await?
} else {
client
.list_store_runs()
.await
.map_err(|err| ToolError::from_anyhow(&err))?
};
let page = filter_sort_and_page_runs(runs, &raw, status.as_deref())?;
Ok(SearchRunsResult {
runs: page.runs.iter().map(search_run_summary_result).collect(),
next_cursor: page.next_cursor,
})
}
fn search_run_summary_result(run: &Run) -> SearchRunSummaryResult {
let RunSummaryResult {
run_id,
workflow_name,
workflow_slug,
status,
archived,
created_at,
started_at,
completed_at,
labels,
source_directory,
repo_origin_url,
goal,
} = common::run_summary_result(run);
let (goal_preview, goal_truncated) = goal_preview(&goal);
SearchRunSummaryResult {
run_id,
workflow_name,
workflow_slug,
status,
archived,
created_at,
started_at,
completed_at,
labels,
source_directory,
repo_origin_url,
goal_preview,
goal_truncated,
}
}
fn goal_preview(goal: &str) -> (String, bool) {
let mut chars = goal.chars();
let mut preview = chars
.by_ref()
.take(SEARCH_GOAL_PREVIEW_CHARS)
.collect::<String>();
let truncated = chars.next().is_some();
if truncated {
preview.push_str("...");
}
(preview, truncated)
}
struct RunSearchPage {
runs: Vec<Run>,
next_cursor: Option<String>,
}
fn filter_sort_and_page_runs(
mut runs: Vec<Run>,
raw: &FabroRunSearchParams,
status: Option<&[RunStatusKind]>,
) -> ToolResult<RunSearchPage> {
if let Some(workflow) = raw.workflow.as_deref() {
runs.retain(|run| {
run.workflow.name == workflow || run.workflow.slug.as_deref() == Some(workflow)
});
}
if let Some(labels) = raw.labels.as_ref() {
runs.retain(|run| {
labels
.iter()
.all(|(key, value)| run.labels.get(key) == Some(value))
});
}
if let Some(status) = status {
runs.retain(|run| {
status
.iter()
.any(|status| *status == run.lifecycle.status.kind())
});
}
let archived = raw.archived.unwrap_or(false);
runs.retain(|run| run.lifecycle.archived == archived);
if let Some(created_after) = raw.created_after.as_deref() {
let cutoff = common::parse_datetime_filter("created_after", created_after)?;
runs.retain(|run| run.timestamps.created_at >= cutoff);
}
if let Some(created_before) = raw.created_before.as_deref() {
let cutoff = common::parse_datetime_filter("created_before", created_before)?;
runs.retain(|run| run.timestamps.created_at <= cutoff);
}
runs.sort_by(|a, b| {
let a_sort_time = a.timestamps.started_at.unwrap_or(a.timestamps.created_at);
let b_sort_time = b.timestamps.started_at.unwrap_or(b.timestamps.created_at);
b_sort_time.cmp(&a_sort_time).then_with(|| b.id.cmp(&a.id))
});
if let Some(after) = raw.after.as_deref() {
if let Some(position) = runs.iter().position(|run| run.id.to_string() == after) {
runs = runs.into_iter().skip(position + 1).collect();
}
}
let first = raw.first.unwrap_or(20).min(100);
let has_more = runs.len() > first;
let page = runs.into_iter().take(first).collect::<Vec<_>>();
let next_cursor = has_more
.then(|| page.last().map(|run| run.id.to_string()))
.flatten();
Ok(RunSearchPage {
runs: page,
next_cursor,
})
}
pub(crate) fn search_runs_text(result: &SearchRunsResult) -> String {
format!("found {} Fabro run(s)", result.runs.len())
}
async fn resolve_requested_runs(client: &Arc<Client>, run_ids: &[String]) -> ToolResult<Vec<Run>> {
let runs = try_join_all(run_ids.iter().map(|run_id| {
let client = Arc::clone(client);
async move {
client
.resolve_run(run_id)
.await
.map_err(|err| ToolError::from_anyhow(&err))
}
}))
.await?;
let mut unique = HashMap::new();
for run in runs {
unique.entry(run.id).or_insert(run);
}
Ok(unique.into_values().collect())
}
#[cfg(test)]
mod tests {
use std::collections::HashMap;
use chrono::{TimeZone, Utc};
use fabro_types::{RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef};
use super::*;
#[test]
fn cursor_is_applied_after_filters() {
let matching_newer = run("01KRBZW5C00000000000000001", "keep", 30);
let unrelated_cursor = run("01KRBZW4DW0000000000000002", "skip", 20);
let matching_older = run("01KRBZW3EF0000000000000003", "keep", 10);
let result = filter_sort_and_page_runs(
vec![
matching_older.clone(),
unrelated_cursor.clone(),
matching_newer.clone(),
],
&FabroRunSearchParams {
run_ids: None,
workflow: None,
labels: Some(HashMap::from([("group".to_string(), "keep".to_string())])),
status: None,
archived: None,
created_after: None,
created_before: None,
first: Some(10),
after: Some(unrelated_cursor.id.to_string()),
},
None,
)
.expect("filtering should succeed");
let ids = result.runs.iter().map(|run| run.id).collect::<Vec<_>>();
assert_eq!(ids, vec![matching_newer.id, matching_older.id]);
}
#[test]
fn omitted_archived_filter_hides_archived_runs_by_default() {
let active = run("01KRBZW5C00000000000000001", "keep", 30);
let archived = archived_run("01KRBZW4DW0000000000000002", "keep", 20);
let result = filter_sort_and_page_runs(
vec![archived.clone(), active.clone()],
&FabroRunSearchParams {
run_ids: None,
workflow: None,
labels: None,
status: None,
archived: None,
created_after: None,
created_before: None,
first: Some(10),
after: None,
},
None,
)
.expect("filtering should succeed");
let ids = result.runs.iter().map(|run| run.id).collect::<Vec<_>>();
assert_eq!(ids, vec![active.id]);
}
#[test]
fn search_summary_uses_bounded_goal_preview() {
let mut run = run("01KRBZW5C00000000000000001", "keep", 30);
run.goal = format!("{}tail-marker", "a".repeat(300));
let summary = search_run_summary_result(&run);
assert!(summary.goal_truncated);
assert!(summary.goal_preview.len() < run.goal.len());
assert!(!summary.goal_preview.contains("tail-marker"));
}
fn run(id: &str, group: &str, seconds: u32) -> Run {
run_with_archived(id, group, seconds, false)
}
fn archived_run(id: &str, group: &str, seconds: u32) -> Run {
run_with_archived(id, group, seconds, true)
}
fn run_with_archived(id: &str, group: &str, seconds: u32, archived: bool) -> Run {
let created_at = Utc.with_ymd_and_hms(2026, 5, 11, 12, 0, seconds).unwrap();
Run {
id: id.parse().expect("test run id should parse"),
title: "test".to_string(),
goal: "test".to_string(),
workflow: WorkflowRef {
slug: Some("simple".to_string()),
name: "Simple".to_string(),
},
automation: None,
repository: None,
created_by: None,
origin: RunOrigin::default(),
labels: HashMap::from([("group".to_string(), group.to_string())]),
lifecycle: RunLifecycle {
status: RunStatus::Submitted,
pending_control: None,
queue_position: None,
error: None,
archived,
archived_at: None,
},
sandbox: None,
models: Vec::new(),
source_directory: None,
timestamps: RunTimestamps {
created_at,
started_at: None,
last_event_at: None,
completed_at: None,
duration_ms: None,
elapsed_secs: None,
},
billing: None,
diff: None,
pull_request: None,
current_question: None,
superseded_by: None,
links: RunLinks { web: None },
}
}
}

View file

@ -0,0 +1,171 @@
use std::path::PathBuf;
use std::sync::Arc;
use anyhow::Result;
use fabro_client::Client;
use rmcp::handler::server::router::tool::ToolRouter;
use rmcp::handler::server::wrapper::Parameters;
use rmcp::model::{CallToolResult, ServerCapabilities, ServerInfo};
use rmcp::transport::stdio;
use rmcp::{ErrorData, ServerHandler, serve_server, tool, tool_handler, tool_router};
use tokio::sync::OnceCell;
use crate::{FabroMcpServerSettings, run_tools};
#[derive(Clone)]
pub(crate) struct FabroMcpServer {
settings: Arc<FabroMcpServerSettings>,
client: Arc<OnceCell<Arc<Client>>>,
cwd: PathBuf,
tool_router: ToolRouter<Self>,
}
pub async fn start(settings: FabroMcpServerSettings) -> Result<()> {
let server = FabroMcpServer::new(Arc::new(settings));
let service = serve_server(server, stdio()).await?;
service.waiting().await?;
Ok(())
}
#[tool_handler(router = self.tool_router)]
impl ServerHandler for FabroMcpServer {
fn get_info(&self) -> ServerInfo {
ServerInfo::new(ServerCapabilities::builder().enable_tools().build())
.with_instructions("Use these tools to create, inspect, control, wait for, and read events from Fabro workflow runs.")
}
}
#[tool_router(router = tool_router)]
impl FabroMcpServer {
pub(crate) fn new(settings: Arc<FabroMcpServerSettings>) -> Self {
let cwd = settings.cwd.clone();
Self {
settings,
client: Arc::new(OnceCell::new()),
cwd,
tool_router: Self::tool_router(),
}
}
#[tool(
name = "fabro_run_create",
description = "Create one or more Fabro workflow runs, starting them by default."
)]
async fn fabro_run_create(
&self,
params: Parameters<run_tools::FabroRunCreateParams>,
) -> Result<CallToolResult, ErrorData> {
let params = match run_tools::ValidatedCreateRuns::try_from(params.0) {
Ok(params) => params,
Err(err) => return Ok(run_tools::error_result(err)),
};
let client = match self.client().await {
Ok(client) => client,
Err(err) => return Ok(run_tools::error_result(err)),
};
match run_tools::create_runs(client, &self.cwd, &self.settings.config_path, params).await {
Ok(result) => run_tools::success_result(&result, run_tools::create_runs_text(&result)),
Err(err) => Ok(run_tools::error_result(err)),
}
}
#[tool(
name = "fabro_run_search",
description = "Search Fabro workflow runs by id, workflow, labels, status, archival state, and creation time."
)]
async fn fabro_run_search(
&self,
params: Parameters<run_tools::FabroRunSearchParams>,
) -> Result<CallToolResult, ErrorData> {
let params = match run_tools::ValidatedSearchRuns::try_from(params.0) {
Ok(params) => params,
Err(err) => return Ok(run_tools::error_result(err)),
};
let client = match self.client().await {
Ok(client) => client,
Err(err) => return Ok(run_tools::error_result(err)),
};
match run_tools::search_runs(client, params).await {
Ok(result) => run_tools::success_result(&result, run_tools::search_runs_text(&result)),
Err(err) => Ok(run_tools::error_result(err)),
}
}
#[tool(
name = "fabro_run_interact",
description = "Get, start, message, cancel, archive, unarchive, inspect questions, or answer a Fabro run."
)]
async fn fabro_run_interact(
&self,
params: Parameters<run_tools::FabroRunInteractParams>,
) -> Result<CallToolResult, ErrorData> {
let params = match run_tools::ValidatedInteractRun::try_from(params.0) {
Ok(params) => params,
Err(err) => return Ok(run_tools::error_result(err)),
};
let client = match self.client().await {
Ok(client) => client,
Err(err) => return Ok(run_tools::error_result(err)),
};
match run_tools::interact_run(client, params).await {
Ok(result) => run_tools::success_result(&result, run_tools::interact_run_text(&result)),
Err(err) => Ok(run_tools::error_result(err)),
}
}
#[tool(
name = "fabro_run_gather",
description = "Wait for Fabro runs to reach terminal states, returning current state on timeout."
)]
async fn fabro_run_gather(
&self,
params: Parameters<run_tools::FabroRunGatherParams>,
) -> Result<CallToolResult, ErrorData> {
let params = match run_tools::ValidatedGatherRuns::try_from(params.0) {
Ok(params) => params,
Err(err) => return Ok(run_tools::error_result(err)),
};
let client = match self.client().await {
Ok(client) => client,
Err(err) => return Ok(run_tools::error_result(err)),
};
match run_tools::gather_runs(client, params).await {
Ok(result) => run_tools::success_result(&result, run_tools::gather_runs_text(&result)),
Err(err) => Ok(run_tools::error_result(err)),
}
}
#[tool(
name = "fabro_run_events",
description = "List, inspect, or search stored events for a Fabro workflow run."
)]
async fn fabro_run_events(
&self,
params: Parameters<run_tools::FabroRunEventsParams>,
) -> Result<CallToolResult, ErrorData> {
let params = match run_tools::ValidatedRunEvents::try_from(params.0) {
Ok(params) => params,
Err(err) => return Ok(run_tools::error_result(err)),
};
let client = match self.client().await {
Ok(client) => client,
Err(err) => return Ok(run_tools::error_result(err)),
};
match run_tools::run_events(client, params).await {
Ok(result) => run_tools::success_result(&result, run_tools::run_events_text(&result)),
Err(err) => Ok(run_tools::error_result(err)),
}
}
async fn client(&self) -> Result<Arc<Client>, run_tools::ToolError> {
self.client
.get_or_try_init(|| async {
(self.settings.client_factory)()
.await
.map(Arc::new)
.map_err(|err| run_tools::ToolError::from_anyhow(&err))
})
.await
.map(Arc::clone)
}
}

View file

@ -22,6 +22,8 @@ enum ClientState {
Connecting(Option<PendingTransport>),
/// Handshake complete, ready for tool calls.
Ready(Arc<RunningService<RoleClient, LoggingClientHandler>>),
/// Connection was explicitly closed.
Closed,
}
enum PendingTransport {
@ -51,9 +53,15 @@ impl McpClient {
.stderr(Stdio::piped())
.kill_on_drop(true);
if config.clear_env {
cmd.env_clear();
}
if !env.is_empty() {
cmd.envs(env);
}
if let Some(current_dir) = config.current_dir.as_ref() {
cmd.current_dir(current_dir);
}
#[cfg(unix)]
cmd.process_group(0);
@ -116,6 +124,7 @@ impl McpClient {
.take()
.ok_or_else(|| anyhow!("client already initializing"))?,
ClientState::Ready(_) => return Err(anyhow!("client already initialized")),
ClientState::Closed => return Err(anyhow!("MCP client is shut down")),
};
// Drop the lock before the blocking handshake
@ -243,11 +252,38 @@ impl McpClient {
Ok(result)
}
pub async fn shutdown(self) -> Result<()> {
let service = {
let mut guard = self.state.lock().await;
match std::mem::replace(&mut *guard, ClientState::Closed) {
ClientState::Connecting(_) | ClientState::Closed => None,
ClientState::Ready(service) => Some(service),
}
};
if let Some(service) = service {
match Arc::try_unwrap(service) {
Ok(mut service) => {
service
.close_with_timeout(Duration::from_secs(2))
.await
.context("failed to shut down MCP client")?;
}
Err(service) => {
service.cancellation_token().cancel();
}
}
}
Ok(())
}
async fn service(&self) -> Result<Arc<RunningService<RoleClient, LoggingClientHandler>>> {
let guard = self.state.lock().await;
match &*guard {
ClientState::Ready(service) => Ok(Arc::clone(service)),
ClientState::Connecting(_) => Err(anyhow!("MCP client not initialized")),
ClientState::Closed => Err(anyhow!("MCP client is shut down")),
}
}
}

View file

@ -13,6 +13,8 @@ fn test_server_config() -> McpServerSettings {
command: vec!["python3".into(), test_server],
env: HashMap::new(),
},
current_dir: None,
clear_env: false,
startup_timeout_secs: 10,
tool_timeout_secs: 30,
}
@ -30,6 +32,78 @@ async fn stdio_client_initialize_and_list_tools() {
assert_eq!(tools[0].1, "Echo back the message");
}
#[tokio::test]
#[expect(
clippy::disallowed_methods,
reason = "stdio integration test stages a local process cwd and inherits PATH for python3 lookup"
)]
async fn stdio_client_uses_configured_cwd_and_exact_env() {
let test_server = format!("{}/tests/test_mcp_server.py", env!("CARGO_MANIFEST_DIR"));
let temp_dir = std::env::temp_dir().join(format!(
"fabro-mcp-stdio-{}-{}",
std::process::id(),
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.unwrap()
.as_nanos()
));
std::fs::create_dir(&temp_dir).unwrap();
let canonical_temp_dir = std::fs::canonicalize(&temp_dir).unwrap();
let mut env = HashMap::new();
env.insert(
"PATH".to_string(),
std::env::var("PATH").expect("PATH should be set for python3 lookup"),
);
env.insert("FABRO_MCP_TEST_SENTINEL".to_string(), "fixture".to_string());
let config = McpServerSettings {
name: "test-echo".into(),
transport: McpTransport::Stdio {
command: vec!["python3".into(), test_server],
env,
},
current_dir: Some(canonical_temp_dir.clone()),
clear_env: true,
startup_timeout_secs: 10,
tool_timeout_secs: 30,
};
let client = McpClient::new(&config).unwrap();
client.initialize(config.startup_timeout()).await.unwrap();
let cwd = client
.call_tool(
"echo",
serde_json::json!({"message": "__cwd__"}),
Duration::from_secs(5),
)
.await
.unwrap();
assert_eq!(
call_result_to_string(&cwd).unwrap(),
canonical_temp_dir.display().to_string()
);
let sentinel = client
.call_tool(
"echo",
serde_json::json!({"message": "__env:FABRO_MCP_TEST_SENTINEL__"}),
Duration::from_secs(5),
)
.await
.unwrap();
assert_eq!(call_result_to_string(&sentinel).unwrap(), "fixture");
let home = client
.call_tool(
"echo",
serde_json::json!({"message": "__env:HOME__"}),
Duration::from_secs(5),
)
.await
.unwrap();
assert_eq!(call_result_to_string(&home).unwrap(), "");
client.shutdown().await.unwrap();
std::fs::remove_dir(&temp_dir).unwrap();
}
#[tokio::test]
async fn stdio_client_call_tool_echo() {
let config = test_server_config();

View file

@ -5,6 +5,7 @@ Speaks JSON-RPC 2.0 over stdin/stdout per the MCP specification.
Exposes a single tool: echo(message) -> message.
"""
import json
import os
import sys
SERVER_INFO = {
@ -53,6 +54,11 @@ def handle_request(req):
arguments = params.get("arguments", {})
if tool_name == "echo":
msg = arguments.get("message", "")
if msg == "__cwd__":
msg = os.getcwd()
elif msg.startswith("__env:") and msg.endswith("__"):
key = msg[len("__env:") : -len("__")]
msg = os.environ.get(key, "")
return {
"jsonrpc": "2.0",
"id": req_id,

View file

@ -24,7 +24,7 @@ anyhow.workspace = true
async-trait.workspace = true
thiserror.workspace = true
tokio.workspace = true
tokio-util.workspace = true
tokio-util = { workspace = true, features = ["compat"] }
serde.workspace = true
serde_json.workspace = true
strum.workspace = true

View file

@ -29,7 +29,7 @@ use crate::redact::redact_auth_url;
use crate::sandbox::{optional_timeout, resolve_path};
use crate::{
CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox,
SandboxEvent, SandboxEventCallback, format_lines_numbered, shell_quote,
SandboxEvent, SandboxEventCallback, StdioProcess, format_lines_numbered, shell_quote,
};
const WORKING_DIRECTORY: &str = "/home/daytona/workspace";
@ -1535,6 +1535,18 @@ impl Sandbox for DaytonaSandbox {
})
}
async fn spawn_stdio_process(
&self,
_command: &str,
_working_dir: Option<&str>,
_env_vars: Option<&HashMap<String, String>>,
_cancel_token: Option<CancellationToken>,
) -> crate::Result<StdioProcess> {
Err(crate::Error::message(
"ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet",
))
}
async fn grep(
&self,
pattern: &str,

View file

@ -1,6 +1,7 @@
use std::collections::BTreeMap;
use anyhow::Result;
#[cfg(any(feature = "docker", feature = "daytona"))]
use chrono::{DateTime, Utc};
use fabro_types::{
RunId, RunSandbox, SandboxDetails, SandboxProvider, SandboxResources, SandboxState,
@ -55,6 +56,7 @@ fn local_details(record: &RunSandbox) -> SandboxDetails {
}
}
#[cfg(any(feature = "docker", feature = "daytona"))]
fn parse_rfc3339_utc(value: &str) -> Option<DateTime<Utc>> {
DateTime::parse_from_rfc3339(value)
.ok()

View file

@ -1,7 +1,8 @@
use std::collections::HashMap;
use std::fmt::Write as _;
use std::io::Cursor;
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
use std::time::{Instant, SystemTime, UNIX_EPOCH};
use async_trait::async_trait;
@ -12,23 +13,25 @@ use bollard::container::{
UploadToContainerOptions,
};
use bollard::errors::Error as DockerError;
use bollard::exec::{CreateExecOptions, StartExecResults};
use bollard::exec::{CreateExecOptions, StartExecOptions, StartExecResults};
use bollard::image::CreateImageOptions;
use bollard::models::HostConfig;
use fabro_github::GitHubCredentials;
use fabro_types::{CommandOutputStream, CommandTermination, RunId};
use fabro_util::time::elapsed_ms;
use futures::StreamExt;
use tokio::sync::OnceCell;
use tokio::io::{AsyncWriteExt, duplex};
use tokio::sync::{Mutex as TokioMutex, Notify, OnceCell};
use tokio::{fs, time};
use tokio_util::sync::CancellationToken;
use crate::clone_source::{self, CloneDecision, EmptyWorkspaceReason};
use crate::redact::redact_auth_url;
use crate::sandbox::{optional_timeout, resolve_path};
use crate::sandbox::{StdioProcessControl, optional_timeout, resolve_path};
use crate::{
CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox,
SandboxEvent, SandboxEventCallback, format_lines_numbered, shell_quote,
CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult,
ExecStreamingResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback, StderrCollector,
StdioProcess, StdioProcessHandle, StdioProcessTermination, format_lines_numbered, shell_quote,
};
const WORKING_DIRECTORY: &str = "/workspace";
@ -216,17 +219,15 @@ impl DockerSandbox {
..Default::default()
};
let exec_instance = self
.docker
.create_exec(container_id, exec_opts)
.await
.map_err(|e| crate::Error::context("Failed to create exec", e))?;
let start_result = self
.docker
.start_exec(&exec_instance.id, None)
.await
.map_err(|e| crate::Error::context("Failed to start exec", e))?;
let (exec_id, start_result) = create_and_start_exec(
&self.docker,
container_id,
exec_opts,
None,
"Failed to create exec",
"Failed to start exec",
)
.await?;
let mut stdout = String::new();
let mut stderr = String::new();
@ -250,7 +251,7 @@ impl DockerSandbox {
let inspect = self
.docker
.inspect_exec(&exec_instance.id)
.inspect_exec(&exec_id)
.await
.map_err(|e| crate::Error::context("Failed to inspect exec", e))?;
@ -278,15 +279,15 @@ impl DockerSandbox {
..Default::default()
};
let exec_instance = docker
.create_exec(&container_id, exec_opts)
.await
.map_err(|e| crate::Error::context("Failed to create exec", e))?;
let start_result = docker
.start_exec(&exec_instance.id, None)
.await
.map_err(|e| crate::Error::context("Failed to start exec", e))?;
let (exec_id, start_result) = create_and_start_exec(
&docker,
&container_id,
exec_opts,
None,
"Failed to create exec",
"Failed to start exec",
)
.await?;
let mut stdout = Vec::new();
let mut stderr = Vec::new();
@ -311,7 +312,7 @@ impl DockerSandbox {
}
let inspect = docker
.inspect_exec(&exec_instance.id)
.inspect_exec(&exec_id)
.await
.map_err(|e| crate::Error::context("Failed to inspect exec", e))?;
@ -451,20 +452,7 @@ impl DockerSandbox {
}
async fn request_docker_exec_stop(&self, stop_file: &str) -> crate::Result<()> {
let command = format!("touch {}", shell_quote(stop_file));
let (stdout, stderr, exit_code) = self
.docker_exec(
vec!["/bin/bash".to_string(), "-lc".to_string(), command.clone()],
Some("/"),
None,
)
.await?;
if exit_code != 0 {
return Err(crate::Error::message(format!(
"Failed to request Docker exec stop (exit {exit_code}): {stderr}{stdout}"
)));
}
Ok(())
request_docker_exec_stop_with(&self.docker, self.container_id()?, stop_file).await
}
async fn ensure_image(&self) -> crate::Result<EnsureImageOutcome> {
@ -769,6 +757,7 @@ fn docker_controlled_shell_command(command: &str, stop_file: &str, pid_file: &st
stop_file={stop_file}; \
pid_file={pid_file}; \
user_command={command}; \
exec 3<&0; \
rm -f \"$pid_file\"; \
if [ -e \"$stop_file\" ]; then \
rm -f \"$stop_file\" \"$pid_file\"; \
@ -783,11 +772,12 @@ fi; \
kill -KILL \"-$child\" 2>/dev/null || kill -KILL \"$child\" 2>/dev/null || true; \
) & watcher=$!; \
if command -v setsid >/dev/null 2>&1; then \
setsid /bin/bash -lc \"$user_command\" & \
setsid /bin/bash -lc \"$user_command\" <&3 & \
else \
/bin/bash -lc \"$user_command\" & \
/bin/bash -lc \"$user_command\" <&3 & \
fi; \
child=$!; \
exec 3<&-; \
echo \"$child\" > \"$pid_file\"; \
wait \"$child\"; \
status=$?; \
@ -804,6 +794,213 @@ exit \"$status\"\
)
}
fn docker_stdio_exec_options(
command: String,
working_dir: String,
env: Option<Vec<String>>,
) -> (CreateExecOptions<String>, StartExecOptions) {
(
CreateExecOptions {
attach_stdin: Some(true),
attach_stdout: Some(true),
attach_stderr: Some(true),
tty: Some(false),
cmd: Some(vec!["/bin/bash".to_string(), "-lc".to_string(), command]),
working_dir: Some(working_dir),
env,
..Default::default()
},
StartExecOptions {
detach: false,
tty: false,
output_capacity: None,
},
)
}
async fn create_and_start_exec(
docker: &Docker,
container_id: &str,
exec_options: CreateExecOptions<String>,
start_options: Option<StartExecOptions>,
create_context: &'static str,
start_context: &'static str,
) -> crate::Result<(String, StartExecResults)> {
let exec_instance = docker
.create_exec(container_id, exec_options)
.await
.map_err(|err| crate::Error::context(create_context, err))?;
let exec_id = exec_instance.id;
let start_result = docker
.start_exec(&exec_id, start_options)
.await
.map_err(|err| crate::Error::context(start_context, err))?;
Ok((exec_id, start_result))
}
async fn request_docker_exec_stop_with(
docker: &Docker,
container_id: &str,
stop_file: &str,
) -> crate::Result<()> {
let command = format!("touch {}", shell_quote(stop_file));
let exec_opts = CreateExecOptions {
cmd: Some(vec!["/bin/bash".to_string(), "-lc".to_string(), command]),
attach_stdout: Some(true),
attach_stderr: Some(true),
working_dir: Some("/".to_string()),
..Default::default()
};
let (exec_id, start_result) = create_and_start_exec(
docker,
container_id,
exec_opts,
None,
"Failed to create Docker exec stop request",
"Failed to start Docker exec stop request",
)
.await?;
let mut stdout = String::new();
let mut stderr = String::new();
if let StartExecResults::Attached { mut output, .. } = start_result {
while let Some(chunk) = output.next().await {
match chunk {
Ok(LogOutput::StdOut { message }) => {
stdout.push_str(&String::from_utf8_lossy(&message));
}
Ok(LogOutput::StdErr { message }) => {
stderr.push_str(&String::from_utf8_lossy(&message));
}
Ok(_) => {}
Err(e) => {
return Err(crate::Error::context(
"Error reading stop request output",
e,
));
}
}
}
}
let inspect = docker
.inspect_exec(&exec_id)
.await
.map_err(|e| crate::Error::context("Failed to inspect Docker exec stop request", e))?;
let exit_code = inspect
.exit_code
.and_then(|code| i32::try_from(code).ok())
.unwrap_or(-1);
if exit_code != 0 {
return Err(crate::Error::message(format!(
"Failed to request Docker exec stop (exit {exit_code}): {stderr}{stdout}"
)));
}
Ok(())
}
struct DockerStdioProcessControl {
docker: Docker,
container_id: String,
exec_id: String,
stop_file: String,
state: Arc<DockerStdioProcessState>,
}
#[derive(Default)]
struct DockerStdioProcessState {
stop_requested: AtomicBool,
termination: TokioMutex<Option<StdioProcessTermination>>,
termination_notify: Notify,
}
impl DockerStdioProcessState {
async fn cached_termination(&self) -> Option<StdioProcessTermination> {
*self.termination.lock().await
}
async fn request_stop_once(&self) -> bool {
self.cached_termination().await.is_none()
&& !self.stop_requested.swap(true, Ordering::AcqRel)
}
async fn cache_termination(&self, termination: StdioProcessTermination) {
let mut cached = self.termination.lock().await;
if cached.is_none() {
*cached = Some(termination);
self.termination_notify.notify_waiters();
}
}
async fn wait_for_cached_termination(&self) -> StdioProcessTermination {
loop {
if let Some(termination) = self.cached_termination().await {
return termination;
}
self.termination_notify.notified().await;
}
}
}
#[async_trait]
impl StdioProcessControl for DockerStdioProcessControl {
async fn terminate(&self) -> crate::Result<()> {
if !self.state.request_stop_once().await {
return Ok(());
}
request_docker_exec_stop_with(&self.docker, &self.container_id, &self.stop_file).await?;
Ok(())
}
async fn wait(&self) -> crate::Result<StdioProcessTermination> {
if let Some(termination) = self.state.cached_termination().await {
return Ok(termination);
}
let mut poll_interval = time::interval(std::time::Duration::from_secs(1));
loop {
if let Some(termination) = self.state.cached_termination().await {
return Ok(termination);
}
let inspect = self
.docker
.inspect_exec(&self.exec_id)
.await
.map_err(|e| crate::Error::context("Failed to inspect Docker stdio exec", e))?;
if inspect.running != Some(true) {
let exit_code = inspect.exit_code.and_then(|code| i32::try_from(code).ok());
let termination = StdioProcessTermination::exited(exit_code);
self.state.cache_termination(termination).await;
return Ok(termination);
}
tokio::select! {
termination = self.state.wait_for_cached_termination() => return Ok(termination),
_ = poll_interval.tick() => {}
}
}
}
}
async fn cache_docker_stdio_completion(
docker: Docker,
exec_id: String,
state: Arc<DockerStdioProcessState>,
) {
match docker.inspect_exec(&exec_id).await {
Ok(inspect) if inspect.running != Some(true) => {
let exit_code = inspect.exit_code.and_then(|code| i32::try_from(code).ok());
state
.cache_termination(StdioProcessTermination::exited(exit_code))
.await;
}
Ok(_) => {}
Err(err) => {
tracing::warn!(error = %err, "Failed to inspect completed Docker stdio exec");
}
}
}
fn git_clone_command(clone_url: &str, branch: Option<&str>) -> String {
let mut command = "git -c maintenance.auto=0 -c gc.auto=0 clone".to_string();
if let Some(branch) = branch {
@ -1333,6 +1530,98 @@ impl Sandbox for DockerSandbox {
.await
}
async fn spawn_stdio_process(
&self,
command: &str,
working_dir: Option<&str>,
env_vars: Option<&HashMap<String, String>>,
cancel_token: Option<CancellationToken>,
) -> crate::Result<StdioProcess> {
let effective_dir = working_dir.map_or_else(
|| WORKING_DIRECTORY.to_string(),
Self::resolve_container_path,
);
let env: Option<Vec<String>> =
env_vars.map(|vars| vars.iter().map(|(k, v)| format!("{k}={v}")).collect());
let (stop_file, pid_file) = docker_exec_control_paths();
let controlled_command = docker_controlled_shell_command(command, &stop_file, &pid_file);
let (create_opts, start_opts) =
docker_stdio_exec_options(controlled_command, effective_dir, env);
let container_id = self.container_id()?.to_string();
let (exec_id, start_result) = create_and_start_exec(
&self.docker,
&container_id,
create_opts,
Some(start_opts),
"Failed to create Docker stdio exec",
"Failed to start Docker stdio exec",
)
.await?;
let StartExecResults::Attached { mut output, input } = start_result else {
return Err(crate::Error::message(
"Docker stdio exec started detached unexpectedly",
));
};
let stderr_collector = StderrCollector::new(DEFAULT_EXEC_OUTPUT_TAIL_BYTES);
let stderr_for_output = stderr_collector.clone();
let (mut stdout_writer, stdout_reader) = duplex(64 * 1024);
let state = Arc::new(DockerStdioProcessState::default());
let state_for_output = Arc::clone(&state);
let docker_for_output = self.docker.clone();
let exec_id_for_output = exec_id.clone();
tokio::spawn(async move {
while let Some(chunk) = output.next().await {
match chunk {
Ok(LogOutput::StdOut { message }) => {
if let Err(err) = stdout_writer.write_all(&message).await {
tracing::warn!(error = %err, "Failed to forward Docker stdio stdout");
break;
}
}
Ok(LogOutput::StdErr { message }) => {
stderr_for_output.push(&message).await;
}
Ok(_) => {}
Err(err) => {
let message = format!("Docker stdio output stream error: {err}");
stderr_for_output.push(message.as_bytes()).await;
break;
}
}
}
cache_docker_stdio_completion(docker_for_output, exec_id_for_output, state_for_output)
.await;
});
let handle = StdioProcessHandle::new(DockerStdioProcessControl {
docker: self.docker.clone(),
container_id,
exec_id,
stop_file,
state,
});
if let Some(token) = cancel_token {
let handle_for_cancel = handle.clone();
tokio::spawn(async move {
token.cancelled().await;
if let Err(err) = handle_for_cancel.terminate().await {
tracing::warn!(error = %err, "Failed to terminate cancelled Docker stdio exec");
}
});
}
Ok(StdioProcess {
stdin: input,
stdout: Box::pin(stdout_reader),
stderr: stderr_collector,
handle,
})
}
async fn read_file(
&self,
path: &str,
@ -1666,8 +1955,10 @@ mod tests {
reason = "unit test reads an in-memory tar entry synchronously"
)]
use std::io::Read as _;
use std::process::Stdio;
use std::time::Duration;
use tokio::io::AsyncWriteExt as _;
use tokio::process::Command;
use super::*;
@ -1744,6 +2035,33 @@ mod tests {
);
}
#[test]
fn stdio_exec_options_attach_streams_without_tty() {
let (create, start) = docker_stdio_exec_options(
"python fake_agent.py".to_string(),
WORKING_DIRECTORY.to_string(),
Some(vec!["MODE=test".to_string()]),
);
assert_eq!(create.attach_stdin, Some(true));
assert_eq!(create.attach_stdout, Some(true));
assert_eq!(create.attach_stderr, Some(true));
assert_eq!(create.tty, Some(false));
assert_eq!(create.working_dir.as_deref(), Some(WORKING_DIRECTORY));
assert_eq!(create.env, Some(vec!["MODE=test".to_string()]));
assert_eq!(
create.cmd,
Some(vec![
"/bin/bash".to_string(),
"-lc".to_string(),
"python fake_agent.py".to_string()
])
);
assert!(!start.detach);
assert!(!start.tty);
assert_eq!(start.output_capacity, None);
}
#[tokio::test]
async fn controlled_shell_command_honors_stop_requested_before_pid_file_exists() {
let tempdir = tempfile::tempdir().expect("tempdir should be created");
@ -1788,6 +2106,59 @@ mod tests {
);
}
#[tokio::test]
async fn controlled_shell_command_preserves_stdin_for_user_command() {
let tempdir = tempfile::tempdir().expect("tempdir should be created");
let stop_file = tempdir.path().join("stop");
let pid_file = tempdir.path().join("pid");
let stop_file = stop_file.to_string_lossy().into_owned();
let pid_file = pid_file.to_string_lossy().into_owned();
let command = docker_controlled_shell_command("cat", &stop_file, &pid_file);
let mut child = Command::new("/bin/bash")
.arg("-lc")
.arg(command)
.stdin(Stdio::piped())
.stdout(Stdio::piped())
.kill_on_drop(true)
.spawn()
.expect("controlled shell command should spawn");
let mut stdin = child
.stdin
.take()
.expect("controlled shell command stdin should be piped");
stdin
.write_all(b"abc\n")
.await
.expect("stdin should be written");
drop(stdin);
let output = time::timeout(Duration::from_secs(5), child.wait_with_output())
.await
.expect("controlled shell command should not hang")
.expect("controlled shell command should run");
assert!(
output.status.success(),
"controlled shell command should exit successfully: {output:?}"
);
assert_eq!(output.stdout, b"abc\n");
}
#[tokio::test]
async fn docker_stdio_process_state_does_not_cache_cancelled_on_stop_request() {
let state = DockerStdioProcessState::default();
assert!(state.request_stop_once().await);
assert_eq!(state.cached_termination().await, None);
assert!(!state.request_stop_once().await);
let termination = StdioProcessTermination::exited(Some(143));
state.cache_termination(termination).await;
assert_eq!(state.cached_termination().await, Some(termination));
assert!(!state.request_stop_once().await);
}
#[tokio::test]
async fn controlled_shell_command_skips_user_command_when_stop_already_requested() {
let tempdir = tempfile::tempdir().expect("tempdir should be created");

View file

@ -41,7 +41,8 @@ pub use reconnect::{reconnect, reconnect_for_run, reconnect_for_run_with_callbac
pub use sandbox::{
CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult,
ExecStreamingResult, GitRunInfo, GitSetupIntent, GrepOptions, Sandbox, SandboxEvent,
SandboxEventCallback, format_lines_numbered, git_push_via_exec, redacted_output_tail,
SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle,
StdioProcessTermination, format_lines_numbered, git_push_via_exec, redacted_output_tail,
setup_git_via_exec, shell_quote,
};
pub use sandbox_spec::SandboxSpec;

View file

@ -7,14 +7,16 @@ use fabro_types::{CommandOutputStream, CommandTermination};
use fabro_util::time::elapsed_ms;
use tokio::io::{AsyncRead, AsyncReadExt};
use tokio::process::{Child, Command};
use tokio::sync::watch;
use tokio::task::spawn_blocking;
use tokio::{fs, time};
use tokio_util::sync::CancellationToken;
use crate::sandbox::optional_timeout;
use crate::sandbox::{StdioProcessControl, optional_timeout};
use crate::{
CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox,
SandboxEvent, SandboxEventCallback, format_lines_numbered,
CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult,
ExecStreamingResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback, StderrCollector,
StdioProcess, StdioProcessHandle, StdioProcessTermination, format_lines_numbered,
};
pub struct LocalSandbox {
@ -128,6 +130,34 @@ fn process_env_vars() -> Vec<(String, String)> {
std::env::vars().collect()
}
#[derive(Debug, Clone, Copy)]
enum ExplicitEnvPolicy {
FilterSensitive,
TrustCaller,
}
fn filtered_env_vars(
env_vars: Option<&std::collections::HashMap<String, String>>,
explicit_policy: ExplicitEnvPolicy,
) -> Vec<(String, String)> {
let mut filtered_env: Vec<(String, String)> = process_env_vars()
.into_iter()
.filter(|(key, _)| !LocalSandbox::should_filter_env_var(key))
.collect();
if let Some(extra) = env_vars {
for (key, value) in extra {
if matches!(explicit_policy, ExplicitEnvPolicy::TrustCaller)
|| !LocalSandbox::should_filter_env_var(key)
{
filtered_env.push((key.clone(), value.clone()));
}
}
}
filtered_env
}
async fn drain_pipe<R>(mut pipe: Option<R>, stream: CommandOutputStream) -> String
where
R: AsyncRead + Unpin,
@ -141,6 +171,77 @@ where
buf
}
type LocalStdioOutcome = Result<StdioProcessTermination, String>;
struct LocalStdioProcessControl {
terminate_tx: watch::Sender<bool>,
termination_rx: watch::Receiver<Option<LocalStdioOutcome>>,
}
impl LocalStdioProcessControl {
fn new(mut child: Child) -> Self {
let (terminate_tx, mut terminate_rx) = watch::channel(false);
let (termination_tx, termination_rx) = watch::channel(None);
tokio::spawn(async move {
let outcome = tokio::select! {
status = child.wait() => {
status
.map(|status| StdioProcessTermination::exited(status.code()))
.map_err(|err| format!("Failed to wait for stdio process: {err}"))
}
changed = terminate_rx.changed() => {
if changed.is_err() || !*terminate_rx.borrow() {
child.wait()
.await
.map(|status| StdioProcessTermination::exited(status.code()))
.map_err(|err| format!("Failed to wait for stdio process: {err}"))
} else {
sigterm_then_kill(&mut child).await;
Ok(StdioProcessTermination::cancelled())
}
}
};
let _ = termination_tx.send(Some(outcome));
});
Self {
terminate_tx,
termination_rx,
}
}
async fn wait_for_termination(&self) -> crate::Result<StdioProcessTermination> {
let mut termination_rx = self.termination_rx.clone();
loop {
if let Some(outcome) = termination_rx.borrow().clone() {
return outcome.map_err(crate::Error::message);
}
termination_rx.changed().await.map_err(|_| {
crate::Error::message(
"stdio process supervisor stopped before reporting termination",
)
})?;
}
}
}
#[async_trait]
impl StdioProcessControl for LocalStdioProcessControl {
async fn terminate(&self) -> crate::Result<()> {
if self.termination_rx.borrow().is_some() {
return Ok(());
}
self.terminate_tx.send_replace(true);
self.wait_for_termination().await.map(|_| ())
}
async fn wait(&self) -> crate::Result<StdioProcessTermination> {
self.wait_for_termination().await
}
}
#[async_trait]
impl Sandbox for LocalSandbox {
async fn read_file(
@ -252,18 +353,7 @@ impl Sandbox for LocalSandbox {
) -> crate::Result<ExecResult> {
let start = Instant::now();
let mut filtered_env: Vec<(String, String)> = process_env_vars()
.into_iter()
.filter(|(key, _)| !Self::should_filter_env_var(key))
.collect();
if let Some(extra) = env_vars {
for (k, v) in extra {
if !Self::should_filter_env_var(k) {
filtered_env.push((k.clone(), v.clone()));
}
}
}
let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::FilterSensitive);
let effective_dir =
working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from);
@ -340,18 +430,7 @@ impl Sandbox for LocalSandbox {
) -> crate::Result<ExecStreamingResult> {
let start = Instant::now();
let mut filtered_env: Vec<(String, String)> = process_env_vars()
.into_iter()
.filter(|(key, _)| !Self::should_filter_env_var(key))
.collect();
if let Some(extra) = env_vars {
for (k, v) in extra {
if !Self::should_filter_env_var(k) {
filtered_env.push((k.clone(), v.clone()));
}
}
}
let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::FilterSensitive);
let effective_dir =
working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from);
@ -424,6 +503,71 @@ impl Sandbox for LocalSandbox {
})
}
async fn spawn_stdio_process(
&self,
command: &str,
working_dir: Option<&str>,
env_vars: Option<&std::collections::HashMap<String, String>>,
cancel_token: Option<CancellationToken>,
) -> crate::Result<StdioProcess> {
let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::TrustCaller);
let effective_dir =
working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from);
let mut cmd = Command::new("/bin/bash");
cmd.arg("-lc")
.arg(format!("exec {command}"))
.current_dir(&effective_dir)
.env_clear()
.envs(filtered_env)
.stdin(std::process::Stdio::piped())
.stdout(std::process::Stdio::piped())
.stderr(std::process::Stdio::piped());
#[cfg(unix)]
fabro_proc::pre_exec_setpgid(cmd.as_std_mut());
let mut child = cmd
.spawn()
.map_err(|e| crate::Error::context("Failed to spawn stdio process", e))?;
let stdin = child
.stdin
.take()
.ok_or_else(|| crate::Error::message("Failed to open stdio process stdin"))?;
let stdout = child
.stdout
.take()
.ok_or_else(|| crate::Error::message("Failed to open stdio process stdout"))?;
let stderr = child
.stderr
.take()
.ok_or_else(|| crate::Error::message("Failed to open stdio process stderr"))?;
let stderr_collector = StderrCollector::new(DEFAULT_EXEC_OUTPUT_TAIL_BYTES);
stderr_collector.spawn_reader(stderr);
let handle = StdioProcessHandle::new(LocalStdioProcessControl::new(child));
if let Some(token) = cancel_token {
let handle_for_cancel = handle.clone();
tokio::spawn(async move {
token.cancelled().await;
if let Err(err) = handle_for_cancel.terminate().await {
tracing::warn!(error = %err, "Failed to terminate cancelled stdio process");
}
});
}
Ok(StdioProcess {
stdin: Box::pin(stdin),
stdout: Box::pin(stdout),
stderr: stderr_collector,
handle,
})
}
async fn grep(
&self,
pattern: &str,
@ -740,7 +884,7 @@ mod tests {
use std::pin::Pin;
use std::task::{Context as TaskContext, Poll};
use tokio::io::ReadBuf;
use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader, ReadBuf};
use super::*;
@ -876,6 +1020,58 @@ mod tests {
std::fs::remove_dir_all(&dir).unwrap();
}
#[tokio::test]
async fn stdio_process_round_trips_lines() {
let dir = temp_dir();
let sandbox = LocalSandbox::new(dir.clone());
let process = sandbox
.spawn_stdio_process(
"python3 -u -c 'import sys; [print(line.strip()[::-1], flush=True) for line in sys.stdin]'",
None,
None,
None,
)
.await
.unwrap();
let mut stdin = process.stdin;
let mut stdout = BufReader::new(process.stdout);
stdin.write_all(b"abc\n").await.unwrap();
stdin.flush().await.unwrap();
let mut line = String::new();
stdout.read_line(&mut line).await.unwrap();
assert_eq!(line.trim_end(), "cba");
process.handle.terminate().await.unwrap();
std::fs::remove_dir_all(&dir).unwrap();
}
#[tokio::test]
async fn stdio_process_forwards_explicit_provider_credentials() {
let dir = temp_dir();
let sandbox = LocalSandbox::new(dir.clone());
let env = HashMap::from([("OPENAI_API_KEY".to_string(), "test-key".to_string())]);
let process = sandbox
.spawn_stdio_process(
"python3 -u -c 'import os; print(os.environ.get(\"OPENAI_API_KEY\", \"missing\"), flush=True)'",
None,
Some(&env),
None,
)
.await
.unwrap();
let mut stdout = BufReader::new(process.stdout);
let mut line = String::new();
stdout.read_line(&mut line).await.unwrap();
assert_eq!(line.trim_end(), "test-key");
process.handle.wait().await.unwrap();
std::fs::remove_dir_all(&dir).unwrap();
}
#[tokio::test]
async fn exec_command_exit_code() {
let dir = temp_dir();

View file

@ -277,4 +277,22 @@ mod tests {
assert!(result.is_ok());
}
#[tokio::test]
async fn stdio_process_forwards_to_inner_sandbox() {
let mock = Arc::new(MockSandbox::linux());
let env = ReadBeforeWriteSandbox::new(mock.clone());
env.spawn_stdio_process("python fake_agent.py", Some("/work/sub"), None, None)
.await
.unwrap();
assert_eq!(
*mock.captured_command.lock().unwrap(),
Some("python fake_agent.py".to_string())
);
assert_eq!(*mock.captured_working_dirs.lock().unwrap(), vec![Some(
"/work/sub".to_string()
)]);
}
}

View file

@ -9,6 +9,9 @@ use std::time::Duration;
use async_trait::async_trait;
use fabro_types::{CommandOutputStream, CommandTermination};
use serde::{Deserialize, Serialize};
use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite};
use tokio::sync::Mutex as TokioMutex;
use tokio::task::JoinHandle;
use tokio::time;
use tokio_util::sync::CancellationToken;
@ -121,6 +124,18 @@ macro_rules! delegate_sandbox {
.await
}
async fn spawn_stdio_process(
&self,
command: &str,
working_dir: Option<&str>,
env_vars: Option<&std::collections::HashMap<String, String>>,
cancel_token: Option<tokio_util::sync::CancellationToken>,
) -> $crate::Result<$crate::StdioProcess> {
self.$field
.spawn_stdio_process(command, working_dir, env_vars, cancel_token)
.await
}
async fn glob(&self, pattern: &str, path: Option<&str>) -> $crate::Result<Vec<String>> {
self.$field.glob(pattern, path).await
}
@ -666,6 +681,114 @@ pub type CommandOutputCallback = Arc<
+ Sync,
>;
pub struct StdioProcess {
pub stdin: Pin<Box<dyn AsyncWrite + Send>>,
pub stdout: Pin<Box<dyn AsyncRead + Send>>,
pub stderr: StderrCollector,
pub handle: StdioProcessHandle,
}
#[derive(Debug, Clone)]
pub struct StderrCollector {
inner: Arc<TokioMutex<Vec<u8>>>,
max_bytes: usize,
}
impl StderrCollector {
#[must_use]
pub fn new(max_bytes: usize) -> Self {
Self {
inner: Arc::new(TokioMutex::new(Vec::new())),
max_bytes,
}
}
pub async fn push(&self, bytes: &[u8]) {
let mut tail = self.inner.lock().await;
tail.extend_from_slice(bytes);
if tail.len() > self.max_bytes {
let excess = tail.len() - self.max_bytes;
tail.drain(..excess);
}
}
pub async fn tail_string(&self) -> String {
let tail = self.inner.lock().await;
String::from_utf8_lossy(&tail).into_owned()
}
pub fn spawn_reader<R>(&self, mut reader: R) -> JoinHandle<()>
where
R: AsyncRead + Unpin + Send + 'static,
{
let collector = self.clone();
tokio::spawn(async move {
let mut buf = [0_u8; 8192];
loop {
match reader.read(&mut buf).await {
Ok(0) => return,
Ok(read) => collector.push(&buf[..read]).await,
Err(err) => {
tracing::warn!(error = %err, "Failed to read stdio process stderr");
return;
}
}
}
})
}
}
#[derive(Clone)]
pub struct StdioProcessHandle {
control: Arc<dyn StdioProcessControl>,
}
impl StdioProcessHandle {
pub(crate) fn new(control: impl StdioProcessControl + 'static) -> Self {
Self {
control: Arc::new(control),
}
}
pub async fn terminate(&self) -> crate::Result<()> {
self.control.terminate().await
}
pub async fn wait(&self) -> crate::Result<StdioProcessTermination> {
self.control.wait().await
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct StdioProcessTermination {
pub termination: CommandTermination,
pub exit_code: Option<i32>,
}
impl StdioProcessTermination {
#[must_use]
pub fn exited(exit_code: Option<i32>) -> Self {
Self {
termination: CommandTermination::Exited,
exit_code,
}
}
#[must_use]
pub fn cancelled() -> Self {
Self {
termination: CommandTermination::Cancelled,
exit_code: None,
}
}
}
#[async_trait]
pub(crate) trait StdioProcessControl: Send + Sync {
async fn terminate(&self) -> crate::Result<()>;
async fn wait(&self) -> crate::Result<StdioProcessTermination>;
}
#[derive(Debug, Clone)]
pub struct DirEntry {
pub name: String,
@ -752,6 +875,19 @@ pub trait Sandbox: Send + Sync {
live_streaming: false,
})
}
async fn spawn_stdio_process(
&self,
_command: &str,
_working_dir: Option<&str>,
_env_vars: Option<&HashMap<String, String>>,
_cancel_token: Option<CancellationToken>,
) -> crate::Result<StdioProcess> {
Err(crate::Error::message(
"ACP backend requires bidirectional stdio; this sandbox provider does not support it",
))
}
async fn grep(
&self,
pattern: &str,

Some files were not shown because too many files have changed in this diff Show more