diff --git a/.claude/skills/changelog/watermark b/.claude/skills/changelog/watermark index 09876abd5..fee3a56da 100644 --- a/.claude/skills/changelog/watermark +++ b/.claude/skills/changelog/watermark @@ -1 +1 @@ -9a3ab8bbba2e71d72ff7703a68e9892e8ce94d8d +a19f6dd03a2ed2b690161476b562462590d78790 diff --git a/.claude/skills/docs/watermark b/.claude/skills/docs/watermark index 578b5ffc5..7b4161ec4 100644 --- a/.claude/skills/docs/watermark +++ b/.claude/skills/docs/watermark @@ -1 +1 @@ -2f10ee39afe6bd9a1c1df987da52fb82c91b422e +aea71a13e5a9c0c276aff04ccc5e2b71e86ce6d6 diff --git a/.config/nextest.toml b/.config/nextest.toml index fe7f4f778..2d5880982 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -15,6 +15,10 @@ leak-timeout = "500ms" filter = "package(fabro-server) & test(all_spec_routes_are_routable)" slow-timeout = { period = "15s", terminate-after = 4 } + [[profile.default.overrides]] + filter = "package(fabro-devcontainer) & test(resolve_features_integration)" + slow-timeout = { period = "10s", terminate-after = 3 } + [[profile.default.overrides]] filter = "package(fabro-workflow)" slow-timeout = { period = "2s", terminate-after = 3 } diff --git a/.fabro/project.toml b/.fabro/project.toml index b050d03d9..3650b5985 100644 --- a/.fabro/project.toml +++ b/.fabro/project.toml @@ -11,7 +11,7 @@ auto_stop_interval = 30 repo = "fabro-sh/fabro" [run.sandbox.daytona.snapshot] -name = "fabro-v8" +name = "fabro-v9" cpu = 8 memory = "16GB" disk = "20GB" @@ -20,6 +20,8 @@ FROM ubuntu:24.04 RUN apt-get update && apt-get install -y --no-install-recommends \ curl git ca-certificates build-essential pkg-config libssl-dev unzip python3 \ + xvfb xfce4 xfce4-terminal x11vnc novnc dbus-x11 \ + libx11-6 libxrandr2 libxext6 libxrender1 libxfixes3 libxss1 libxtst6 libxi6 \ && rm -rf /var/lib/apt/lists/* # GitHub CLI diff --git a/.fabro/workflows/daytona-medium/workflow.fabro b/.fabro/workflows/daytona-medium/workflow.fabro new file mode 100644 index 000000000..1b801c763 --- /dev/null +++ b/.fabro/workflows/daytona-medium/workflow.fabro @@ -0,0 +1,11 @@ +digraph DaytonaMedium { + graph [goal="Verify the Daytona daytona-medium sandbox starts with standard tooling", retry_target=exit] + rankdir=LR + + start [shape=Mdiamond, label="Start"] + exit [shape=Msquare, label="Exit"] + + inspect [label="Inspect Sandbox", shape=parallelogram, goal_gate=true, script="set -e\nprintf 'cwd: '; pwd\nprintf 'user: '; whoami\nprintf 'git: '; git --version\nif command -v python3 >/dev/null; then printf 'python: '; python3 --version; else echo 'python: not installed'; fi\nif command -v node >/dev/null; then printf 'node: '; node --version; else echo 'node: not installed'; fi\nprintf 'top-level files:\\n'; ls -la | sed -n '1,40p'"] + + start -> inspect -> exit +} diff --git a/.fabro/workflows/daytona-medium/workflow.toml b/.fabro/workflows/daytona-medium/workflow.toml new file mode 100644 index 000000000..ec86fbdba --- /dev/null +++ b/.fabro/workflows/daytona-medium/workflow.toml @@ -0,0 +1,10 @@ +_version = 1 + +[workflow] +graph = "workflow.fabro" + +[run.sandbox] +provider = "daytona" + +[run.sandbox.daytona.snapshot] +name = "daytona-medium" diff --git a/Cargo.lock b/Cargo.lock index 655be86cb..fa3ced888 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -58,6 +58,72 @@ dependencies = [ "subtle", ] +[[package]] +name = "agent-client-protocol" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2af62fb84df2af0f933d8f5fd78b843fa5eb0ec5a48fa1b528c41951d0bbe36c" +dependencies = [ + "agent-client-protocol-derive", + "agent-client-protocol-schema", + "anyhow", + "futures", + "futures-concurrency", + "jsonrpcmsg", + "rmcp", + "rustc-hash", + "schemars 1.2.1", + "serde", + "serde_json", + "thiserror 2.0.18", + "tokio", + "tokio-util", + "tracing", + "uuid", +] + +[[package]] +name = "agent-client-protocol-derive" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce42c2d3c048c12897eef2e577dfff1e3355c632c9f1625cc953b9df48b44631" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + +[[package]] +name = "agent-client-protocol-schema" +version = "0.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49bae57dad1c28a362fbdcf7bab0583316a02b45a70792109fced55780a3b63c" +dependencies = [ + "anyhow", + "derive_more", + "schemars 1.2.1", + "serde", + "serde_json", + "serde_with", + "strum", + "tracing", +] + +[[package]] +name = "agent-client-protocol-tokio" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e1572b219f22c4b3be0f20f934c8b6f1d1457126ce72923c4f6608f96153b65" +dependencies = [ + "agent-client-protocol", + "futures", + "serde", + "serde_json", + "shell-words", + "tokio", + "tokio-util", +] + [[package]] name = "ahash" version = "0.8.12" @@ -594,6 +660,15 @@ version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dc0b364ead1874514c8c2855ab558056ebfeb775653e7ae45ff72f28f8f3166c" +[[package]] +name = "bs58" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +dependencies = [ + "tinyvec", +] + [[package]] name = "bstr" version = "1.12.1" @@ -1088,16 +1163,6 @@ dependencies = [ "darling_macro 0.14.4", ] -[[package]] -name = "darling" -version = "0.21.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9cdf337090841a411e2a7f3deb9187445851f91b309c0c0a29e05f74a00a48c0" -dependencies = [ - "darling_core 0.21.3", - "darling_macro 0.21.3", -] - [[package]] name = "darling" version = "0.23.0" @@ -1122,20 +1187,6 @@ dependencies = [ "syn 1.0.109", ] -[[package]] -name = "darling_core" -version = "0.21.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1247195ecd7e3c85f83c8d2a366e4210d588e802133e1e355180a9870b517ea4" -dependencies = [ - "fnv", - "ident_case", - "proc-macro2", - "quote", - "strsim 0.11.1", - "syn 2.0.117", -] - [[package]] name = "darling_core" version = "0.23.0" @@ -1160,17 +1211,6 @@ dependencies = [ "syn 1.0.109", ] -[[package]] -name = "darling_macro" -version = "0.21.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d38308df82d1080de0afee5d069fa14b0326a88c14f15c5ccda35b4a6c414c81" -dependencies = [ - "darling_core 0.21.3", - "quote", - "syn 2.0.117", -] - [[package]] name = "darling_macro" version = "0.23.0" @@ -1309,6 +1349,7 @@ dependencies = [ "quote", "rustc_version", "syn 2.0.117", + "unicode-xid", ] [[package]] @@ -1537,9 +1578,31 @@ dependencies = [ "libc", ] +[[package]] +name = "fabro-acp" +version = "0.231.0-nightly.1" +dependencies = [ + "agent-client-protocol", + "agent-client-protocol-tokio", + "bytes", + "fabro-model", + "fabro-sandbox", + "fabro-types", + "fabro-util", + "futures", + "serde", + "serde_json", + "tempfile", + "thiserror 2.0.18", + "tokio", + "tokio-util", + "tracing", + "uuid", +] + [[package]] name = "fabro-agent" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -1578,7 +1641,7 @@ dependencies = [ [[package]] name = "fabro-api" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "chrono", "fabro-config", @@ -1599,7 +1662,7 @@ dependencies = [ [[package]] name = "fabro-auth" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -1623,11 +1686,11 @@ dependencies = [ [[package]] name = "fabro-build-support" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" [[package]] name = "fabro-checkpoint" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "chrono", "fabro-config", @@ -1643,7 +1706,7 @@ dependencies = [ [[package]] name = "fabro-cli" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "assert_cmd", @@ -1661,6 +1724,7 @@ dependencies = [ "dialoguer", "dirs", "dotenvy", + "fabro-acp", "fabro-agent", "fabro-api", "fabro-auth", @@ -1678,7 +1742,9 @@ dependencies = [ "fabro-interview", "fabro-llm", "fabro-macros", + "fabro-manifest", "fabro-mcp", + "fabro-mcp-server", "fabro-model", "fabro-oauth", "fabro-proc", @@ -1741,7 +1807,7 @@ dependencies = [ [[package]] name = "fabro-client" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "bytes", @@ -1770,7 +1836,7 @@ dependencies = [ [[package]] name = "fabro-config" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "chrono", @@ -1797,7 +1863,7 @@ dependencies = [ [[package]] name = "fabro-core" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "async-trait", "fabro-types", @@ -1812,7 +1878,7 @@ dependencies = [ [[package]] name = "fabro-dev" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "assert_cmd", @@ -1831,7 +1897,7 @@ dependencies = [ [[package]] name = "fabro-devcontainer" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "fabro-http", "fabro-static", @@ -1848,7 +1914,7 @@ dependencies = [ [[package]] name = "fabro-dump" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "bytes", @@ -1862,7 +1928,7 @@ dependencies = [ [[package]] name = "fabro-github" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "base64", @@ -1884,7 +1950,7 @@ dependencies = [ [[package]] name = "fabro-graphviz" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "fabro-types", @@ -1898,7 +1964,7 @@ dependencies = [ [[package]] name = "fabro-hooks" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "async-trait", "fabro-agent", @@ -1922,7 +1988,7 @@ dependencies = [ [[package]] name = "fabro-http" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "fabro-static", "http", @@ -1932,7 +1998,7 @@ dependencies = [ [[package]] name = "fabro-install" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "base64", @@ -1947,7 +2013,7 @@ dependencies = [ [[package]] name = "fabro-interview" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "async-trait", "dialoguer", @@ -1962,7 +2028,7 @@ dependencies = [ [[package]] name = "fabro-llm" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -1994,7 +2060,7 @@ dependencies = [ [[package]] name = "fabro-macros" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "clap", "fabro-options-metadata", @@ -2003,9 +2069,27 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "fabro-manifest" +version = "0.231.0-nightly.1" +dependencies = [ + "anyhow", + "fabro-api", + "fabro-config", + "fabro-github", + "fabro-graphviz", + "fabro-template", + "fabro-types", + "fabro-workflow", + "git2", + "temp-env", + "tempfile", + "toml 0.8.23", +] + [[package]] name = "fabro-mcp" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "fabro-config", @@ -2019,9 +2103,31 @@ dependencies = [ "tracing", ] +[[package]] +name = "fabro-mcp-server" +version = "0.231.0-nightly.1" +dependencies = [ + "anyhow", + "chrono", + "fabro-api", + "fabro-client", + "fabro-config", + "fabro-manifest", + "fabro-server", + "fabro-types", + "fabro-util", + "futures", + "rmcp", + "schemars 1.2.1", + "serde", + "serde_json", + "tokio", + "toml 0.8.23", +] + [[package]] name = "fabro-model" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "fabro-static", "insta", @@ -2032,7 +2138,7 @@ dependencies = [ [[package]] name = "fabro-oauth" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "axum", @@ -2054,7 +2160,7 @@ dependencies = [ [[package]] name = "fabro-options-metadata" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "serde", "serde_json", @@ -2062,7 +2168,7 @@ dependencies = [ [[package]] name = "fabro-proc" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "cc", "libc", @@ -2071,7 +2177,7 @@ dependencies = [ [[package]] name = "fabro-redact" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "aho-corasick", "ref-cast", @@ -2087,7 +2193,7 @@ dependencies = [ [[package]] name = "fabro-sandbox" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -2130,7 +2236,7 @@ dependencies = [ [[package]] name = "fabro-server" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -2155,6 +2261,7 @@ dependencies = [ "fabro-interview", "fabro-llm", "fabro-macros", + "fabro-manifest", "fabro-model", "fabro-proc", "fabro-redact", @@ -2211,7 +2318,7 @@ dependencies = [ [[package]] name = "fabro-slack" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "fabro-http", "fabro-interview", @@ -2232,18 +2339,18 @@ dependencies = [ [[package]] name = "fabro-spa" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "rust-embed", ] [[package]] name = "fabro-static" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" [[package]] name = "fabro-store" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "async-trait", "bytes", @@ -2270,7 +2377,7 @@ dependencies = [ [[package]] name = "fabro-telemetry" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "base64", @@ -2296,7 +2403,7 @@ dependencies = [ [[package]] name = "fabro-template" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "fabro-util", @@ -2308,7 +2415,7 @@ dependencies = [ [[package]] name = "fabro-test" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "assert_cmd", "axum", @@ -2331,7 +2438,7 @@ dependencies = [ [[package]] name = "fabro-tracker" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-trait", @@ -2345,7 +2452,7 @@ dependencies = [ [[package]] name = "fabro-types" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "chrono", "clap", @@ -2366,7 +2473,7 @@ dependencies = [ [[package]] name = "fabro-util" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "console 0.15.11", @@ -2386,17 +2493,18 @@ dependencies = [ [[package]] name = "fabro-validate" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "fabro-graphviz", "fabro-model", + "fabro-types", "serde", "thiserror 2.0.18", ] [[package]] name = "fabro-vault" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "chrono", "fabro-types", @@ -2408,7 +2516,7 @@ dependencies = [ [[package]] name = "fabro-workflow" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "assert_cmd", @@ -2417,6 +2525,7 @@ dependencies = [ "bytes", "chrono", "dirs", + "fabro-acp", "fabro-agent", "fabro-auth", "fabro-checkpoint", @@ -2530,6 +2639,12 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +[[package]] +name = "fixedbitset" +version = "0.5.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d674e81391d1e1ab681a28d99df07927c6d4aa5b027d7da16ba32d1d21ecd99" + [[package]] name = "flatbuffers" version = "25.12.19" @@ -2800,6 +2915,19 @@ dependencies = [ "futures-sink", ] +[[package]] +name = "futures-concurrency" +version = "7.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "175cd8cca9e1d45b87f18ffa75088f2099e3c4fe5e2f83e42de112560bea8ea6" +dependencies = [ + "fixedbitset", + "futures-core", + "futures-lite", + "pin-project", + "smallvec", +] + [[package]] name = "futures-core" version = "0.3.32" @@ -2823,6 +2951,19 @@ version = "0.3.32" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +[[package]] +name = "futures-lite" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f78e10609fe0e0b3f4157ffab1876319b5b0db102a2c60dc4626306dc46b44ad" +dependencies = [ + "fastrand", + "futures-core", + "futures-io", + "parking", + "pin-project-lite", +] + [[package]] name = "futures-macro" version = "0.3.32" @@ -3689,6 +3830,16 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "jsonrpcmsg" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6d833a15225c779251e13929203518c2ff26e2fe0f322d584b213f4f4dad37bd" +dependencies = [ + "serde", + "serde_json", +] + [[package]] name = "jsonschema" version = "0.42.2" @@ -5587,6 +5738,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2231b2c085b371c01bc90c0e6c1cab8834711b6394533375bdbf870b0166d419" dependencies = [ "async-trait", + "base64", "chrono", "futures", "http", @@ -6134,11 +6286,12 @@ dependencies = [ [[package]] name = "serde_with" -version = "3.17.0" +version = "3.20.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "381b283ce7bc6b476d903296fb59d0d36633652b633b27f64db4fb46dcbfc3b9" +checksum = "e72c1c2cb7b223fafb600a619537a871c2818583d619401b785e7c0b746ccde2" dependencies = [ "base64", + "bs58", "chrono", "hex", "indexmap 1.9.3", @@ -6153,11 +6306,11 @@ dependencies = [ [[package]] name = "serde_with_macros" -version = "3.17.0" +version = "3.20.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6d4e30573c8cb306ed6ab1dca8423eec9a463ea0e155f45399455e0368b27e0" +checksum = "b90c488738ecb4fb0262f41f43bc40efc5868d9fb744319ddf5f5317f417bfac" dependencies = [ - "darling 0.21.3", + "darling 0.23.0", "proc-macro2", "quote", "syn 2.0.117", @@ -6888,6 +7041,7 @@ checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" dependencies = [ "bytes", "futures-core", + "futures-io", "futures-sink", "futures-util", "hashbrown 0.15.5", @@ -7140,7 +7294,7 @@ dependencies = [ [[package]] name = "twin-github" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "axum", "base64", @@ -7159,7 +7313,7 @@ dependencies = [ [[package]] name = "twin-openai" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" dependencies = [ "anyhow", "async-stream", diff --git a/Cargo.toml b/Cargo.toml index 17fc5e550..15aad22e9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -5,10 +5,12 @@ resolver = "2" [workspace.package] edition = "2021" -version = "0.230.0-nightly.0" +version = "0.231.0-nightly.1" license = "MIT" [workspace.dependencies] +agent-client-protocol = { version = "0.11.1", features = ["unstable_session_usage"] } +agent-client-protocol-tokio = "0.11.1" anyhow = "1" axum = { version = "0.8" } axum-extra = { version = "0.10", features = ["cookie-private"] } diff --git a/docs/internal/events.md b/docs/internal/events.md index ce4c37d3f..f548196d6 100644 --- a/docs/internal/events.md +++ b/docs/internal/events.md @@ -1952,6 +1952,8 @@ Emitted when an image or snapshot ensure step fails. ## CLI ensure events +These legacy events may appear in older run logs. Current CLI backend runs do not emit them because Fabro no longer installs or prepares provider CLIs at stage runtime. + ### `cli.ensure.started` ```json diff --git a/docs/internal/fabro-event-schema-v2-concrete-shape.md b/docs/internal/fabro-event-schema-v2-concrete-shape.md index b4032a495..feee1a2b5 100644 --- a/docs/internal/fabro-event-schema-v2-concrete-shape.md +++ b/docs/internal/fabro-event-schema-v2-concrete-shape.md @@ -401,7 +401,7 @@ V2 keeps the current durable family surface broadly intact. - `sandbox.*` - `setup.*` -- `cli.ensure.*` +- `cli.ensure.*` (legacy only) - `command.*` - `agent.cli.*` - `devcontainer.*` diff --git a/docs/internal/mcp-server-qa-test-plan.md b/docs/internal/mcp-server-qa-test-plan.md new file mode 100644 index 000000000..c3f6a6b5e --- /dev/null +++ b/docs/internal/mcp-server-qa-test-plan.md @@ -0,0 +1,246 @@ +# Fabro MCP Server — QA Test Plan + +One-time manual QA pass for the 5 tools exposed by `fabro-mcp-server`. Source of truth: `lib/crates/fabro-mcp-server/src/run_tools/`. + +This plan is **not** a template for adding automated test coverage — it exists to drive a single hands-on sweep against a real running server. Tick boxes as scenarios pass; add notes inline for failures or surprising behavior. Open bugs/PRs for issues found; do not port these scenarios into the Rust test suite. + +## Findings rollup + +Live list of bugs and notable observations surfaced during the sweep. Each entry links back to the scenario where it was found. + +### Bugs / mismatches +None currently open. + +### Rechecked / no longer open +- **C4 — `inputs` schema/runtime mismatch**: fixed by narrowing MCP input values to scalar JSON (`string`, `boolean`, `integer`, `number`) and rejecting arrays/objects locally with scalar-only errors. Re-tested on 2026-05-11 against `127.0.0.1:32276`; `tools/list` now advertises scalar-only `inputs.additionalProperties`. +- **C5 — Misleading null-input error message**: fixed. Re-tested on 2026-05-11; null now returns ``input `maybe` cannot be null; use a string, boolean, or number``. +- **I7 / I9 — Misleading "Run not found." on terminal runs**: fixed on 2026-05-11 in the server API layer. `message`/steer against a durable terminal run that no longer has a live managed engine now returns `409` with `run_not_steerable`; `cancel` returns `409` with `Run is already terminal and cannot be cancelled.` True missing runs still return `404`. +- **I10 — Archived runs not filtered from default search**: fixed on 2026-05-11 by aligning MCP search with the HTTP API. `fabro_run_search` now hides archived runs when `archived` is omitted, while `archived=true` still searches archived runs explicitly. +- **I15 / I16 — yes/no answer flow**: re-tested on 2026-05-11 against `fabro server` `0.230.0-nightly.0` at `127.0.0.1:32276`. `answer=true` and `answer=false` both submit successfully for the bundled `interview` workflow's first `yes_no` question. `true` advanced the run to the next `confirmation` question. +- **I22 — numeric answer local validation**: re-tested on 2026-05-11 against the same server. `answer=42` now returns `unsupported answer value: 42; expected boolean, string, or object` from the MCP layer before reaching the API. +- **Section 2 side observation — Search payloads include full `goal` text**: fixed on 2026-05-11. `fabro_run_search` now returns bounded `goal_preview` plus `goal_truncated` instead of the full `goal`, keeping list responses compact while preserving full summaries on other run interactions. +- **X6 — Cursor/filter ordering**: simplified on 2026-05-11 by applying search filters before sorting and applying the `after` cursor. This prevents unrelated runs outside the filtered result set from trimming the page. Pagination is explicitly not snapshot-isolated; a new matching run inserted before the cursor during traversal appears when the client starts a new search. + +### UX / polish +- **C12 — `cwd` errors don't distinguish "directory missing" from "workflow not in directory"**: both return `workflow not found: `. +- **S9 (bonus) — Undocumented date format**: error message reveals `YYYY-MM-DD` is accepted alongside RFC3339, but the schema only says RFC3339. +- **S17 — `run_ids` accepts more than IDs**: error message reveals it also matches ID prefixes and workflow names. Either rename the field or document. +- **E4 — Events `search` is whole-envelope substring match**: search includes embedded payloads (workflow definitions, settings, sandbox dockerfile, etc.), so a search like `query="list_prs"` legitimately matches the `run.created` event because that event embeds the workflow JSON. Easy to misinterpret. Consider documenting or scoping search to event body only. + +### Nice-to-haves +- **C16 — Helpful error**: unknown workflow lists available workflows. Keep. + +## Pre-flight (all tools) + +- [ ] **P1** Server unreachable — stop `fabro server`, call any tool, expect a clear connection-error message (not a panic, not a hang). +- [ ] **P2** Schema discovery — list tools through an MCP client; verify each tool has a complete JSON schema and the documented `anyOf` for `AnswerValue`. + +--- + +## 1. `fabro_run_create` + +Source: `run_tools/create.rs:124` + +### Happy path +- [x] **C1** Create one run from an existing workflow (e.g. `gh-list`); default `start=true` → expect `started=true`, `status` in `{queued, starting, running}`. — **PASS**. `status=queued`. +- [x] **C2** Create with `start=false` → expect `started=false`, `status=submitted`. — **PASS**. Run `01KRC4MP2NEQS9GJDE9FJ0EECH` kept as fixture for I3. +- [x] **C3** Batch create 5 runs in one call → all return; result preserves array order. — **PASS**. ULIDs monotonically increasing. + +### Inputs / manifest +- [x] **C4** Pass `inputs` with string / number / boolean / nested object / array → **PASS** after 2026-05-11 recheck. Scalar values are accepted. Arrays and objects are rejected locally with scalar-only errors, and the MCP schema now advertises scalar-only `inputs` values. +- [x] **C5** `inputs` containing `null` → **PASS** after 2026-05-11 recheck. Returns ``input `maybe` cannot be null; use a string, boolean, or number``. +- [x] **C6** `labels={"team": "qa"}` round-trip via search. — **PASS**. All 5 C3 runs returned with labels intact. +- [x] **C7** Optional flags: `goal`, `model+provider`, `sandbox`, `preserve_sandbox+auto_approve+dry_run`. — **PASS** all accepted; `goal` override round-tripped via search. +- [x] **C8** Custom `run_id`: valid ULID accepted (`01KRC500000000C8TEST00000A`); wrong length → `invalid length`; invalid Crockford char (e.g. `U`) → `invalid character`. — **PASS**. + +### `cwd` +- [x] **C9** Omit `cwd` → uses base CWD. — **PASS** (covered by every prior scenario). +- [x] **C10** `cwd` to repo root resolves workflow. — **PASS**. +- [x] **C11** `cwd=/tmp` (no `.fabro/workflows`) → `workflow not found: gh-list`. — **PASS**. +- [x] **C12** `cwd=/this/path/does/not/exist/xyz123` → same generic `workflow not found: gh-list`. — **PASS but note**: error doesn't distinguish "directory missing" from "workflow not in directory". Minor UX gap. + +### Validation +- [x] **C13** Empty `runs: []` → `runs must contain at least 1 item(s)`. — **PASS**. +- [x] **C14** 51 entries → `runs must contain no more than 50 item(s)`. — **PASS**. +- [x] **C15** Missing required `workflow` → MCP layer `-32602: missing field 'workflow'`. — **PASS**. +- [x] **C16** Unknown workflow slug → `Unknown workflow 'X'\n\nAvailable workflows: ...`. — **PASS** (very helpful — lists available workflows). + +### Failure semantics +- [x] **C17** Invalid sandbox name → `failed to resolve manifest settings: run.sandbox.provider: invalid value - unknown sandbox provider: this-sandbox-does-not-exist`. — **PASS**. Error raised at manifest-resolve time before any run record is created (no orphaned submitted run). + +--- + +## 2. `fabro_run_search` + +Source: `run_tools/search.rs:75` + +### Happy path +- [x] **S1** No params → returns up to 20 runs, sorted by `started_at OR created_at` desc. — **PASS**. Mixed-timestamp ordering correct (succeeded run at pos 8 sorts by its `started_at` between two `created_at`-only runs). +- [x] **S2** `first=5` → exactly 5; `next_cursor` is the last run's ID. — **PASS**. +- [x] **S3** `first=100` → all 17 runs, `next_cursor=null`. — **PASS**. +- [x] **S4** Cursor follow-through: page 1 IDs `[A, B]`, page 2 with `after=B` returns `[C, D]`. No overlap. — **PASS**. Note: cursors are run IDs, not opaque tokens. + +### Filters +- [x] **S5** `workflow="smoke"` (slug) and `workflow="Smoke"` (name) both match same run. — **PASS**. +- [x] **S6** `status=["succeeded"]` → 4; `["failed","dead"]` → 1; `["submitted"]` → 5. — **PASS**. +- [x] **S7** Labels round-trip. — **PASS** (verified via C6). +- [x] **S8** `archived=false` → all unarchived runs; `archived=true` → `[]` (no archived runs yet). Re-verify after I10. — **PARTIAL** (no archived fixtures yet). +- [x] **S9** `created_after`/`created_before` (RFC3339) bound results correctly; tight window `17:00–18:00` returns only old runs. — **PASS**. **Bonus**: error message reveals `YYYY-MM-DD` is also accepted — undocumented in the schema. +- [x] **S10** `run_ids=[A,B,A]` → 2 deduped runs. — **PASS**. +- [x] **S11** Combined `workflow + status + labels + archived` → returns exactly the 5 batch=c3 runs. — **PASS**. + +### Validation +- [x] **S12** `first=101` → `first must be <= 100`. — **PASS**. +- [x] **S13** `run_ids=[]` → `run_ids must contain at least 1 item(s)`. — **PASS**. +- [x] **S14** `run_ids` length 101 → `run_ids must contain no more than 100 item(s)`. — **PASS**. +- [x] **S15** `status=["bogus"]` → `unknown run status 'bogus'`. — **PASS**. +- [x] **S16** `created_after="not-a-date"` → `created_after must be RFC3339 or YYYY-MM-DD: input contains invalid characters`. — **PASS**. + +### Edge cases +- [x] **S17** Non-existent ID in `run_ids` → `No run found matching '' (tried run ID prefix and workflow name)`. — **PASS** + **finding**: `run_ids` also accepts ID prefixes and workflow names, which is broader than the field name suggests. +- [x] **S18** No matches → `{"runs": [], "next_cursor": null}`. — **PASS**. +- [x] **S19** Bogus `after=` → returns full first page (skip never applies). — **PASS** as documented. + +### Side observation +Search responses include the full `goal` text per run; a single `ImplementPlan` run can add ~30 KB to every search payload. Consider truncating `goal` (or excluding it from list responses) the way events have `max_content_length`. **Logged in Findings.** + +--- + +## 3. `fabro_run_gather` + +Source: `run_tools/gather.rs:56` + +### Happy path +- [x] **G1** Gather 1 already-terminal run → instant return, `timed_out=false`, `elapsed_seconds=0`. — **PASS**. +- [x] **G2** In-flight `gh-list` with `timeout=60, poll=5` → reaches `succeeded`, `timed_out=false`, `elapsed=30`. — **PASS**. +- [x] **G3** In-flight `gh-list` with `timeout=5, poll=5` → `timed_out=true`, `elapsed=5`, run still `starting`. — **PASS**. +- [x] **G4** Mix of 2 terminal + 1 in-flight, `timeout=90, poll=5` → all 3 succeeded, `timed_out=false`, `elapsed=40`. — **PASS**. + +### Validation +- [x] **G5** `run_ids=[]` → `run_ids must contain at least 1 item(s)`. — **PASS**. +- [x] **G6** 51 IDs → `run_ids must contain no more than 50 item(s)`. — **PASS**. +- [x] **G7** `timeout_seconds=601` → `timeout_seconds must be <= 600`. — **PASS**. +- [x] **G8** `poll_interval_seconds=4` → `poll_interval_seconds must be >= 5`. — **PASS**. +- [x] **G9** Omit both → call accepted; terminal run still returns instantly. Default values per source: `timeout=300, poll=15`. — **PASS**. + +### Edge cases +- [x] **G10** Non-existent run ID → `No run found matching '' (tried run ID prefix and workflow name)`. — **PASS** (same fuzzy match as search). +- [x] **G11** Poll cadence: G3 confirms last sleep clamps to deadline (`elapsed=5` exactly with `timeout=5, poll=5`). — **PASS** (inferred from G2/G3 timing). +- [ ] **G12** Run cancelled mid-gather → terminal `failed(status_reason=cancelled)` quickly. — **DEFERRED** to after I8 (cancel). +- [ ] **G13** Run becomes `blocked` — verify gather still waits. — **DEFERRED** to after Section 5 (interview workflow). + +--- + +## 4. `fabro_run_events` + +Source: `run_tools/events.rs:115` + +### Actions +- [x] **E1** `list` no filters → 45 events (`gh-list` has full lifecycle: run.*, sandbox.*, git.*, stage.*, etc.), `next_cursor=46`. — **PASS**. +- [x] **E2** `details` with 2 event_ids → returns exactly those 2 envelopes. — **PASS**. +- [x] **E3** `details` with no `event_ids` → `event_ids is required for details action`. — **PASS**. +- [x] **E4** `search query="list_prs"` → 14 events. Includes `run.created` because it embeds the full workflow definition (which contains the `list_prs` node ID). — **PASS** + **observation**: search ranges over the entire serialized envelope, so big embedded payloads (workflow defs, settings) can produce non-obvious hits. +- [x] **E5** `search` with missing `query` → `query is required for search action`. — **PASS**. + +### Filters +- [x] **E6** `event_types=["stage.started"]` → exactly 4 events (start, list_prs, list_issues, exit). — **PASS**. +- [x] **E7** `categories=["git","sandbox"]` → 12 events all with prefix `git.*` or `sandbox.*`. — **PASS**. +- [x] **E8** `created_after=17:03:10Z` + `created_before=17:03:13Z` → 5 events all timestamped 17:03:12.89x. — **PASS**. +- [x] **E9** Combined `event_types + offset + first` covered by E14. + +### Pagination & direction +- [x] **E10** Page 1 `first=10` → seqs 1–10, `next_cursor=11`. Page 2 `after=11, first=5` → seqs 11–15, `next_cursor=16`. No duplicates; contiguous. — **PASS**. +- [x] **E11** `direction=desc, first=5` → seqs 45, 44, 43, 42, 41; `next_cursor=41` (last seq, no +1 — per the desc branch). — **PASS**. +- [x] **E12** Default direction = asc (E10 confirms). — **PASS**. +- [x] **E13** `direction="weird"` → `direction must be 'asc' or 'desc'`. — **PASS**. +- [x] **E14** `event_types=["stage.started"], offset=2, first=5` → returned 2 events (seqs 29, 39) — correctly skipped the first 2 (15, 19) of the 4 matching. — **PASS**. +- [x] **E15** `limit=3` → 3 events. — **PASS** (alias works). + +### Truncation +- [x] **E16** `stage.completed, first=1, max_content_length=200` → 1 event, `truncated=true`, `event` is a JSON string. — **PASS**. +- [x] **E17** UTF-8 boundary — **VERIFIED via existing unit test** at `events.rs:269-312`. Can't easily reproduce through MCP surface (no multibyte event content in default fixtures). +- [x] **E18** Default `max_content_length=20000` → all 5 events `truncated=false` (including the ~5 KB `run.created`). — **PASS**. + +### Validation +- [x] **E19** `run_id=" "` (whitespace) → `run_id is required`. — **PASS**. +- [x] **E20** `first=201` → `first must be <= 200`. — **PASS**. +- [x] **E21** Non-existent run ID → fuzzy-match error (same as search/gather). — **PASS**. + +--- + +## 5. `fabro_run_interact` + +Source: `run_tools/interact.rs:201` + +### Actions + +#### `get` +- [x] **I1** Returns `{summary, projection}`; projection includes `spec`, `graph`, `status`, `checkpoints`, `pending_interviews`, `stages`, `sandbox`, `conclusion`, etc. — **PASS**. +- [x] **I2** Non-existent run → fuzzy match error. — **PASS**. + +#### `start` +- [x] **I3** Non-started run (from C2) → `start` transitions to `queued`. Second `start` → `an engine process is still running for this run — cannot start`. — **PASS**. + +#### `message` (steer) +- [ ] **I4** Steer a running LLM agent — **DEFERRED** (requires an active LLM agent stage; would burn LLM tokens; can be exercised manually once the answer bug below is resolved). +- [ ] **I5** `interrupt=true` — **DEFERRED** along with I4. +- [x] **I6** Missing `message` → `message is required for action message`. — **PASS**. +- [x] **I7** Message a terminal run → initially returned `Run not found.`. — **FIXED**: durable terminal runs without a live managed engine now return `409 run_not_steerable`; true missing runs remain `404`. + +#### `cancel` +- [x] **I8** Cancel a `gh-list` run during `starting`. Returns summary at request time (status=`starting`). Subsequent `gather` returned terminal `failed` within 5s; `get` projection shows `status: {kind: "failed", reason: "cancelled"}` and `conclusion.failure_reason: "Pipeline cancelled"`. — **PASS** + **observation**: `cancel`'s returned summary is a snapshot at request time, not the eventual terminal status. +- [x] **I9** Cancel an already-terminal run → initially returned `Run not found.`. — **FIXED**: durable terminal runs without a live managed engine now return `409` with `Run is already terminal and cannot be cancelled.`; true missing runs remain `404`. + +#### `archive` / `unarchive` +- [x] **I10** Archive terminal run → `archived=true` in summary; visible via `search archived=true`. — **FIXED**: default search now hides archived runs to match `/api/v1/runs`; `archived=true` still surfaces archived runs explicitly. +- [x] **I11** Unarchive → reverses (`archived=false`). — **PASS**. +- [x] **I12** Archive an active run → `run must be terminal (succeeded, failed, or dead) to archive; current status is starting`. — **PASS** (excellent error). + +#### `get_questions` +- [x] **I13** Terminal run → `questions: []`. — **PASS**. +- [x] **I14** Blocked interview run → returns full question record (id, text, options, question_type, stage, allow_freeform). — **PASS**. + +#### `answer` — `AnswerValue` shapes + +Re-check note: the earlier `yes_no` answer failure did not reproduce against `fabro server` `0.230.0-nightly.0` on `127.0.0.1:32276` (2026-05-11). Boolean answers are accepted for `yes_no` questions, and invalid question/type combinations are rejected by the API as expected. + +- [x] **I15** `answer=true` on the first `yes_no` question → submitted successfully (`submitted=true`) and advanced to the `confirmation` question. — **PASS**. Run `01KRCAQ9AS14KFCW4CXBZQ0CW9`. +- [x] **I16** `answer=false` on a fresh `yes_no` question → submitted successfully (`submitted=true`). — **PASS**. Run `01KRCATZ031CAPEPVB4CNFEE33`. +- [ ] **I17** `answer="some text"` — **NOT RE-TESTED**. Should be tested against a `freeform` question or a question with `allow_freeform=true`; text is not valid for the bundled `yes_no` question. +- [ ] **I18** `answer={"text":"hi"}` — **NOT RE-TESTED**. Same scope as I17. +- [x] **I19** `answer={"option":"Y"}` against the first `yes_no` question → `Answer does not match question type.` — **PASS / expectation corrected**. The MCP layer maps this shape to `selected`, but `server.rs:2670-2710` only accepts `yes`/`no` for `yes_no` and `confirmation`; `selected` belongs to `multiple_choice`. +- [ ] **I20** `answer={"options":[...]}` — **NOT RE-TESTED**. Should be tested against a `multi_select` question; `multi_selected` is not valid for `yes_no`. +- [x] **I21** `answer={"value":"yes"}` → `answer object must contain one of: option, options, text` (local validation). — **PASS**. +- [x] **I22** `answer=42` (number) → `unsupported answer value: 42; expected boolean, string, or object`. — **PASS** (local validation). +- [x] **I23** `answer={"option": 5}` → `answer option must be a string: invalid type: integer '5', expected a string`. — **PASS**. +- [x] **I24** `answer={"options": ["a", 2]}` → `answer options must be strings: invalid type: integer '2', expected a string`. — **PASS**. +- [x] **I25** `action=answer` without `question_id` → `question_id is required for action answer`. — **PASS**. +- [x] **I26** `action=answer` without `answer` → `answer is required for action answer`. — **PASS**. +- [x] **I27** Already-answered question — observed indirectly: the same question_id returned `Question no longer exists or was already answered.` on retry. — **PASS**. + +### Cross-cutting +- [x] **I28** `run_id=" "` → `run_id is required`. — **PASS**. +- [x] **I29** Action enum: `Get` and `get-questions` both rejected with `unknown variant 'X', expected one of: get, start, message, cancel, archive, unarchive, get_questions, answer`. — **PASS**. + +--- + +## 6. End-to-end scenarios (multi-tool) + +- [x] **X1 — Happy lifecycle** `gh-list` create → 35s gather → events filtered to `stage.started/completed` → 8 events for 4 stages (start, list_prs, list_issues, exit). Sequence matches workflow graph. — **PASS**. +- [x] **X2 — Cancel mid-run** Covered by I8: `gh-list` cancel during `starting` → gather returned terminal `failed` in 5s; projection shows `status_reason=cancelled`. — **PASS**. +- [ ] **X3 — Human-in-the-loop** — **PARTIAL**. The earlier yes/no answer blocker is no longer reproduced (I15/I16 now pass), and `gather` returning `timed_out=true` on a `blocked` run **was** verified (G13). Full interview completion remains unverified in this sweep. +- [ ] **X4 — Steering** — **DEFERRED** (requires active LLM agent). +- [x] **X5 — Archive flow** Covered by I10/I11: archive → search with `archived=true` returns it (also returned by default search — see I10 finding). Unarchive reverses. — **PASS** with caveat. +- [x] **X6 — Search/cursor under churn** Page 1 `first=3` → cursor saved. Created new run `01KRC625KG…` mid-flow. Page 2 with original cursor returned 3 older runs; a fresh page 1 placed the new run at position 1. — **ACCEPTED / SIMPLIFIED**. Pagination is not snapshot-isolated; clients that need newly inserted earlier results should restart the search. Code now applies filters before sorting/cursoring so unrelated runs outside the filtered result set do not trim filtered pages. +- [x] **X7 — Events while running** Started `gh-list` run, listed events `desc` immediately (max seq=8), gathered to completion, re-listed (max seq=46). Seq numbers grew monotonically; no early events lost. — **PASS**. +- [x] **X8 — Truncated event recovery** Fetched the `ImplementPlan` `run.created` event (embeds ~30 KB goal) at default `max_content_length=20000` → `truncated:true`, payload returned as a JSON string. — **PASS**. +- [ ] **X9 — Stranger inputs** — Skipped per scope decision. Trivially safe since inputs go through TOML conversion to be stored as values; the MCP layer never opens paths. + +--- + +## 7. Mechanics for the manual sweep + +- **Driver** — run these scenarios through an MCP client (e.g. Claude Code with the `fabro` MCP server configured) against a locally running `fabro server`. +- **Reusable run IDs** — keep a handful of already-terminal runs around (e.g. one `gh-list` succeeded, one failed `implement-plan`) as fixtures for `events`, `gather` (instant-return), `interact.get`, and `archive` scenarios. +- **Server unreachable cases** — stop the API server with the MCP client still connected to exercise error propagation paths. +- **Issue tracking** — file a GitHub issue per defect; link the scenario ID (e.g. `C13`) so this plan and the bugs cross-reference. diff --git a/docs/plans/2026-05-11-add-acp-backend-test-plan.md b/docs/plans/2026-05-11-add-acp-backend-test-plan.md new file mode 100644 index 000000000..3a54ee109 --- /dev/null +++ b/docs/plans/2026-05-11-add-acp-backend-test-plan.md @@ -0,0 +1,326 @@ +# ACP Backend Test Plan + +The accepted testing strategy still holds, with scoped additions from the implementation plan: ACP prompt nodes are explicitly supported, backend validation becomes strict, sandbox stdio is part of the public contract, and ACP events affect run projection, fork replay, and server steerability. These additions do not require paid services or materially change the agreed scope because all high-value ACP checks can run against deterministic fake agents and local/unit harnesses. + +## Harness Requirements + +1. **Fake ACP agent harness** + - What it does: runs a deterministic ACP agent over stdio, records observed JSON-RPC method order, emits configurable `session/update` messages, writes optional files in cwd, responds to permission requests, and simulates cancellation, malformed JSON, early exit, timeout, and stop reasons. + - Exposes: a checked-in fixture binary or script plus crate-local helpers in `fabro-acp::test_support` using `agent-client-protocol` schema types where practical. + - Complexity: medium. It is the main substitute for paid/live ACP agents. + - Tests depending on it: 7, 8, 9, 10, 11, 12, 13, 26, 27. + +2. **Sandbox stdio process harness** + - What it does: exercises `Sandbox::spawn_stdio_process` with a line-oriented subprocess, captures stdout/stderr separately, terminates the process, and validates Docker exec option construction without requiring live Docker. + - Exposes: local sandbox round-trip tests, Docker option-builder/control-wrapper tests, Daytona unsupported-provider assertion, and decorator/test-support forwarding assertions. + - Complexity: medium because Docker stdio is multiplexed and cancellation needs an explicit control path. + - Tests depending on it: 5, 6, 8, 12, 26. + +3. **Workflow ACP runner harness** + - What it does: runs a real Fabro workflow with `backend="acp"` and node-level `acp_command` pointing at the fake ACP agent, then inspects persisted run events/projection through existing CLI workflow helpers. + - Exposes: user-visible `fabro run` result, run events, stage response, `files_touched`, and `provider_used`. + - Complexity: low once the fake ACP agent exists. + - Tests depending on it: 26, 27. + +4. **Server event-state harness extension** + - What it does: uses existing server test fixtures to insert a running run, apply ACP events, and call `POST /runs/{id}/steer`. + - Exposes: HTTP status and JSON error codes through the Axum test router. + - Complexity: low. + - Tests depending on it: 23, 24. + +## Test Plan + +1. **Existing CLI backend behavior remains intact** + - Type: regression + - Disposition: existing + - Harness: existing `fabro-workflow` router/CLI tests from the accepted strategy. + - Preconditions: current repository before ACP changes; no ACP-specific code required. + - Actions: run `ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_uses_cli_for_backend_attr) | test(router_uses_api_by_default) | test(backend_router_delegates_to_cli_for_cli_node) | test(backend_router_delegates_to_api_for_normal_node) | test(backend_router_delegates_to_cli_for_backend_attr) | test(full_pipeline_with_cli_backend_node) | test(stylesheet_backend_property_routes_to_cli) | test(cli_backend_run_writes_prompt_and_calls_exec) | test(cli_backend_run_with_codex_provider) | test(parse_real_codex_ndjson)'`. + - Expected outcome: all tests pass; `backend="cli"` still routes agent nodes to CLI, default routing remains API, stylesheet `backend: cli` still works, CLI output parsing remains unchanged. Source of truth: user request to keep `api`, `cli`, and `acp` as three backends for now; implementation plan User-Visible Behavior for legacy CLI compatibility. + - Interactions: router, CLI backend, stylesheet import, sandbox command execution, CLI event emission. + +2. **Existing stdio JSON-RPC precedent remains intact** + - Type: regression + - Disposition: existing + - Harness: existing `fabro-mcp` stdio integration tests. + - Preconditions: Python is available; no live MCP service required. + - Actions: run `ulimit -n 4096 && cargo nextest run -p fabro-mcp -E 'test(stdio_client_initialize_and_list_tools) | test(stdio_client_call_tool_echo) | test(connection_manager_stdio_roundtrip)'`. + - Expected outcome: all tests pass; Fabro can still spawn a stdio JSON-RPC collaborator, initialize it, list capabilities, and call it. Source of truth: accepted strategy listed these as relevant stdio precedent. + - Interactions: child process stdio, JSON-RPC framing, local subprocess lifecycle. + +3. **ACP default command mapping matches provider families** + - Type: unit + - Disposition: new + - Harness: `fabro-acp` command mapping tests. + - Preconditions: `fabro-acp` crate exists with `agent-client-protocol-tokio = 0.11.1`. + - Actions: call `default_acp_command` for Anthropic, OpenAI, Kimi, Zai, Minimax, Inception, OpenAI-compatible, and Gemini. + - Expected outcome: Anthropic maps to `npx -y @zed-industries/claude-code-acp@latest`; OpenAI-compatible family maps to `npx -y @zed-industries/codex-acp@latest`; Gemini maps to `npx -y -- @google/gemini-cli@latest --experimental-acp`. Source of truth: implementation plan User-Visible Behavior default ACP command mapping. + - Interactions: provider enum coverage and ACP Tokio parser defaults. + +4. **ACP command overrides are parsed as stdio commands, not raw shell** + - Type: boundary + - Disposition: new + - Harness: `fabro-acp` command parsing tests using `agent_client_protocol_tokio::AcpAgent::from_str`. + - Preconditions: no sandbox required. + - Actions: resolve `acp_command` values for a shell-word command, a blank string, a JSON stdio config with args/env, and a non-stdio JSON config. + - Expected outcome: shell-word and JSON stdio commands expose parsed program/args/env; blank overrides fail with `acp_command must not be empty`; HTTP/SSE configs fail with `only stdio ACP commands are supported`; rendered sandbox command uses parsed parts with shell quoting. Source of truth: implementation plan command override contract and shell quoting invariant in `AGENTS.md`. + - Interactions: ACP Tokio parser, command rendering, env merge inputs. + +5. **Local sandbox stdio round-trips without a PTY** + - Type: integration + - Disposition: new + - Harness: `fabro-sandbox` local stdio process harness. + - Preconditions: temp local sandbox workspace; Python or a POSIX shell command available. + - Actions: spawn a line-oriented process with `spawn_stdio_process`, write `abc\n` to stdin, read one stdout line, then terminate and wait. + - Expected outcome: stdout returns the transformed line, stderr remains separately collectible, and `terminate()` completes without leaking the process. Source of truth: implementation plan Contracts And Invariants requiring sandbox-backed, bidirectional, non-PTY stdio. + - Interactions: local process groups, env filtering, async IO, cancellation cleanup. + +6. **Sandbox providers preserve or reject ACP stdio capability correctly** + - Type: invariant + - Disposition: new + - Harness: `fabro-sandbox` provider/decorator tests. + - Preconditions: local sandbox, read/write guard, worktree/decorator wrappers, test-support sandbox, and Daytona provider stub are available. + - Actions: call `spawn_stdio_process` through each wrapper around a supporting sandbox; call it on Daytona; construct Docker exec stdio options. + - Expected outcome: wrappers forward to the inner sandbox; Daytona returns `ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet`; Docker create/start options attach stdin/stdout/stderr and set `tty=false`; Docker termination uses the stop-file/control path. Source of truth: implementation plan provider support and PTY corruption risk. + - Interactions: decorator macro, worktree path resolution, Docker option builder, Daytona provider boundary. + +7. **ACP lifecycle initializes, creates a session, sends a prompt, and aggregates text** + - Type: integration + - Disposition: new + - Harness: `fabro-acp` fake ACP agent harness. + - Preconditions: fake ACP agent configured to emit two text `agent_message_chunk` updates and return `stopReason: "end_turn"`. + - Actions: call `run_acp_turn` with a prompt and cwd. + - Expected outcome: fake agent observes `initialize`, `session/new`, `session/prompt` in order; result text is the concatenation of text chunks; stop reason is `EndTurn`. Source of truth: ACP initialization/session/prompt docs and docs.rs quick-start lifecycle. + - Interactions: official ACP SDK client, sandbox stdio transport, JSON-RPC ordering. + +8. **ACP runs inside the active sandbox and sees the workflow cwd** + - Type: integration + - Disposition: new + - Harness: `fabro-acp` fake agent plus local sandbox stdio. + - Preconditions: temp sandbox workspace; fake agent writes `hello.txt` in its cwd during `session/prompt`. + - Actions: call `run_acp_turn`, then inspect the sandbox workspace for `hello.txt`. + - Expected outcome: file exists inside the sandbox workspace, not the host process cwd; `session/new` cwd matches `sandbox.working_directory()`. Source of truth: implementation plan Contracts And Invariants requiring ACP processes to run inside the active Fabro sandbox. + - Interactions: sandbox cwd resolution, command launch, file mutation visibility. + +9. **ACP permission requests auto-select an allow option** + - Type: integration + - Disposition: new + - Harness: `fabro-acp` fake ACP agent harness. + - Preconditions: fake agent sends `session/request_permission` with `AllowAlways`, `AllowOnce`, and reject options before completing the prompt. + - Actions: call `run_acp_turn` and record the client response. + - Expected outcome: client responds with the `AllowAlways` option id when present, then the turn continues and returns text. Source of truth: implementation plan permission handling contract; ACP supports agent-to-client permission requests. + - Interactions: ACP client request handler, cancellation token state, prompt turn progress. + +10. **ACP cancellation sends session cancel and returns cancellation** + - Type: boundary + - Disposition: new + - Harness: `fabro-acp` fake ACP agent harness. + - Preconditions: fake agent has created a session and is holding `session/prompt` open. + - Actions: start `run_acp_turn`, cancel the token before completion, and let the fake agent record incoming notifications. + - Expected outcome: client sends `session/cancel` for the active session, terminates if the agent does not finish within grace, and returns `AcpError::Cancelled`. If a permission request arrives after cancellation, the response is `RequestPermissionOutcome::Cancelled`. Source of truth: implementation plan cancellation contract and ACP prompt lifecycle stop reasons. + - Interactions: cancel token, JSON-RPC notification, process termination. + +11. **ACP timeout terminates the process and reports timeout** + - Type: boundary + - Disposition: new + - Harness: `fabro-acp` fake ACP agent harness. + - Preconditions: fake agent never responds to `session/prompt`; request timeout is short. + - Actions: call `run_acp_turn`. + - Expected outcome: process is terminated, stderr tail is available if emitted, and error is `AcpError::TimedOut`. Source of truth: implementation plan timeout contract using node timeout like CLI mode. + - Interactions: watchdog activity, process handle termination, stderr collector. + +12. **ACP protocol failures include diagnostic stderr without losing typed errors** + - Type: boundary + - Disposition: new + - Harness: `fabro-acp` fake ACP agent harness. + - Preconditions: fake agents for malformed JSON-RPC and early nonzero exit. + - Actions: call `run_acp_turn` for each failure mode. + - Expected outcome: malformed JSON returns a protocol error; early exit includes exit status and stderr tail; error source chains remain inspectable where applicable. Source of truth: implementation plan malformed/early-exit behavior and error-handling strategy. + - Interactions: ACP SDK error propagation, stderr tail collection, process wait. + +13. **ACP stop reasons map to Fabro backend outcomes** + - Type: boundary + - Disposition: new + - Harness: `fabro-acp` fake agent plus workflow `AgentAcpBackend` adapter tests. + - Preconditions: fake agent can return `EndTurn`, `Refusal`, `Cancelled`, `MaxTokens`, and `MaxTurnRequests`. + - Actions: run an ACP backend turn for each stop reason. + - Expected outcome: `EndTurn` and `Refusal` return text; `Cancelled` maps to `Error::Cancelled`; `MaxTokens` and `MaxTurnRequests` return handler errors containing the stop reason and partial output. Source of truth: implementation plan Stop reason handling. + - Interactions: protocol result mapping, workflow error conversion, event terminal paths. + +14. **ACP backend adapter prepares credentials, env, Node runtime, and changed files** + - Type: integration + - Disposition: new + - Harness: `fabro-workflow` ACP adapter tests with fake credential resolver and fake sandbox. + - Preconditions: node uses `backend="acp"`; fake resolver can provide env vars and login command; sandbox records commands and git status before/after. + - Actions: call `AgentAcpBackend::run`. + - Expected outcome: login command runs before ACP; tool env overlays command env; default `npx` commands trigger Node/npm/npx bootstrap; explicit `acp_command` does not install provider CLIs; `files_touched` excludes pre-existing dirty files and includes new changed/untracked files. Source of truth: implementation plan env preparation, Node bootstrap, and changed-file semantics. + - Interactions: credential resolver, workflow tool env, sandbox exec, Git diff helper. + +15. **ACP one-shot prompt nodes use sandboxed ACP and combine system prompt correctly** + - Type: integration + - Disposition: new + - Harness: `fabro-workflow` `PromptHandler` and `AgentAcpBackend::one_shot` tests. + - Preconditions: prompt node has `backend="acp"`; project memory can produce a system prompt; fake backend captures sandbox pointer and cancellation token. + - Actions: execute the prompt handler. + - Expected outcome: `PromptHandler` passes the active sandbox and run cancel token into `CodergenBackend::one_shot`; ACP one-shot sends `System:\n{system_prompt}\n\nUser:\n{prompt}` when system prompt exists and only the prompt when absent; no host process is used. Source of truth: implementation plan User-Visible Behavior for prompt/one_shot ACP support. + - Interactions: prompt handler, memory discovery, backend trait signature, run services. + +16. **Backend router selects api, cli, and acp explicitly** + - Type: integration + - Disposition: extend + - Harness: `fabro-workflow` router tests. + - Preconditions: router has API, CLI, and ACP test backends with distinguishable responses. + - Actions: run agent nodes with absent backend, `backend="api"`, `backend="cli"`, `backend="acp"`, and `backend="codex"`. + - Expected outcome: absent and `api` use API; `cli` uses CLI; `acp` uses ACP; unknown backend fails with `unsupported LLM backend "codex"; expected one of: api, cli, acp`. Source of truth: implementation plan three-way router selection and strict validation requirement. + - Interactions: node attributes, model fallback, handler errors. + +17. **Prompt router keeps legacy cli one-shot fallback but routes acp to ACP** + - Type: regression + - Disposition: extend + - Harness: `fabro-workflow` router one-shot tests. + - Preconditions: router has API and ACP one-shot test backends. + - Actions: call `one_shot` for prompt nodes with absent backend, `backend="api"`, `backend="cli"`, and `backend="acp"`. + - Expected outcome: absent, `api`, and legacy `cli` prompt nodes use API; `acp` uses ACP. Source of truth: implementation plan compatibility note for prompt nodes with `backend="cli"` and explicit ACP prompt support. + - Interactions: backend routing, prompt handler behavior, backward compatibility. + +18. **Workflow validation accepts only supported backend values** + - Type: boundary + - Disposition: new + - Harness: `fabro-validate` `backend_valid` rule tests and CLI validate coverage if practical. + - Preconditions: graphs with absent backend and with `api`, `cli`, `acp`, and `codex`. + - Actions: run `fabro_validate::validate` against each graph; optionally run `fabro validate` against an invalid fixture. + - Expected outcome: absent, `api`, `cli`, and `acp` have no backend diagnostic; `codex` returns an error diagnostic containing `unsupported LLM backend "codex"; expected one of: api, cli, acp`. Source of truth: implementation plan User-Visible Behavior for unknown backend values. + - Interactions: validation registry, parser, CLI diagnostic rendering. + +19. **Imported workflow placeholders propagate acp_command** + - Type: regression + - Disposition: extend + - Harness: `fabro-workflow` import transform tests. + - Preconditions: host workflow has an import placeholder with `backend="acp"` and `acp_command="python fake_agent.py"`; imported workflow has LLM nodes. + - Actions: run `ImportTransform` and inspect imported node attrs. + - Expected outcome: imported LLM nodes receive `backend="acp"` and the placeholder `acp_command`; unsupported placeholder attributes still poison the placeholder. Source of truth: implementation plan file list and import transform requirement. + - Interactions: graph transform, default attribute propagation, import validation. + +20. **ACP events serialize with stage-scoped metadata** + - Type: integration + - Disposition: new + - Harness: `fabro-workflow` event conversion tests. + - Preconditions: construct `Event::AgentAcpStarted`, `AgentAcpCompleted`, `AgentAcpCancelled`, and `AgentAcpTimedOut` with a `StageScope`. + - Actions: convert each event through `to_run_event`. + - Expected outcome: event names are `agent.acp.started`, `agent.acp.completed`, `agent.acp.cancelled`, and `agent.acp.timed_out`; envelope includes `node_id`, stage id/visit-derived fields, and no prompt/env/credential contents. Source of truth: events strategy and implementation plan ACP event contract. + - Interactions: event naming, stored fields, `fabro-types` event body serde. + +21. **Run projection records ACP provider metadata and terminal output** + - Type: integration + - Disposition: new + - Harness: `fabro-store` run projection tests. + - Preconditions: event sequence has stage start, `agent.acp.started`, terminal ACP event, and stage completion/failure. + - Actions: apply events to `RunProjection`. + - Expected outcome: `stage.provider_used.mode == "acp"` with provider, model, and command; completed output contains aggregated text/stderr payload; cancelled and timed-out terminal events set `CommandTermination::Cancelled` and `CommandTermination::TimedOut`. Source of truth: implementation plan run projection support. + - Interactions: stored event fields, stage lookup by visit, projection terminal data. + +22. **Fork replay preserves ACP stage metadata** + - Type: regression + - Disposition: new + - Harness: `fabro-workflow` fork replay tests. + - Preconditions: source run history includes ACP started/cancelled/timed-out events before a checkpoint. + - Actions: call fork replay filtering or run a lower-level fork projection test. + - Expected outcome: `AgentAcpStarted`, `AgentAcpCancelled`, and `AgentAcpTimedOut` are replayed into the fork projection; `AgentAcpCompleted` follows the existing CLI completed replay policy. Source of truth: implementation plan fork replay requirement. + - Interactions: historical event filtering, forked run projection. + +23. **ACP running stages are not steerable through the server API** + - Type: scenario + - Disposition: new + - Harness: server event-state harness extension. + - Preconditions: a running managed run with worker control channel; no active API-mode agent session; active stage marker has been set by `agent.acp.started`. + - Actions: call `POST /runs/{id}/steer` with a plain steer request and with interrupt+steer. + - Expected outcome: response is `409 CONFLICT` with a clear non-steerable-agent error code/message; no worker control message is enqueued as if an API session might appear. Source of truth: implementation plan server steerability tracking. + - Interactions: run manager event reducer, HTTP handler, worker control queue. + +24. **ACP non-steerable marker clears on all terminal paths** + - Type: invariant + - Disposition: new + - Harness: server event-state harness extension. + - Preconditions: a running managed run with active ACP stage and no active API-mode stage. + - Actions: apply each clearing event independently: `agent.acp.completed`, `agent.acp.cancelled`, `agent.acp.timed_out`, `stage.completed`, and `stage.failed`; then call `POST /runs/{id}/steer` with a plain steer request. + - Expected outcome: plain steer is accepted/buffered after each terminal event because no non-steerable active agent remains. Source of truth: implementation plan server steerability clearing rules. + - Interactions: event reducer backstops, HTTP handler, stage lifecycle. + +25. **Pipeline initialization wires ACP into real workflow handlers** + - Type: integration + - Disposition: new + - Harness: `fabro-workflow` pipeline initialization tests. + - Preconditions: graph contains `backend="acp"` LLM node; credentials are supplied through a stub/env source; dry-run and non-dry-run cases are both available. + - Actions: call `initialize`/`build_registry` and execute or resolve the node through the initialized registry using a fake ACP runner. + - Expected outcome: non-dry-run registry constructs a router with ACP; dry-run still builds no real backend and simulates LLM handlers; ACP does not fall back to host env when a resolver exists. Source of truth: implementation plan pipeline initialization task. + - Interactions: credential source, handler registry, dry-run path. + +26. **Black-box `fabro run` executes an ACP-backed agent workflow** + - Type: scenario + - Disposition: new + - Harness: workflow ACP runner harness in `fabro-cli/tests/it/workflow`. + - Preconditions: temp workflow has an agent node with `backend="acp"`, `provider="openai"`, `model="fake-acp"`, and `acp_command` pointing to the checked-in fake ACP agent; local sandbox is used. + - Actions: run the workflow through the CLI test command, then read run state/events through existing workflow helpers. + - Expected outcome: run succeeds; stage response contains concatenated chunks; `hello.txt` is included in `files_touched`; run projection has `provider_used.mode == "acp"`; `agent.acp.started` and `agent.acp.completed` events are present. Source of truth: user request for first-class `backend="acp"` and implementation plan black-box workflow coverage. + - Interactions: CLI command, parser, validation, pipeline initialization, sandbox stdio, ACP protocol, run store. + +27. **Black-box ACP prompt workflow uses ACP instead of API** + - Type: scenario + - Disposition: new + - Harness: workflow ACP runner harness. + - Preconditions: temp workflow has a prompt/one_shot node with `backend="acp"` and fake ACP command. + - Actions: run the workflow through the CLI test command and inspect stage response/events. + - Expected outcome: prompt node succeeds through ACP, response is fake ACP text, and `agent.acp.*` provider metadata appears; no API-mode `agent.session.activated` event is needed for the prompt. Source of truth: implementation plan User-Visible Behavior for `backend="acp"` on prompt/one_shot nodes. + - Interactions: prompt handler, one-shot routing, pipeline initialization, run projection. + +28. **Documentation examples and backend references include ACP without stale CLI prompt claims** + - Type: regression + - Disposition: extend + - Harness: documentation grep plus existing docs build if normally run in CI. + - Preconditions: docs have been updated. + - Actions: run `rg -n "backend=.*cli|backend: cli|backend.*api|CLI backend|cli mode|ACP" docs/public lib/crates -g '*.md' -g '*.mdx'` and `cd apps/marketing && bun run build` only if the touched docs are built by that package. + - Expected outcome: docs mention valid backend values `api`, `cli`, `acp`; `cli` is described as legacy; ACP sandbox and Daytona limitations are documented; no stale claim remains that prompt nodes use CLI mode. Source of truth: implementation plan documentation task. + - Interactions: public docs, marketing/docs build pipeline. + +29. **Final targeted ACP verification passes** + - Type: invariant + - Disposition: new + - Harness: repository test suites named by the implementation plan. + - Preconditions: all implementation tasks complete. + - Actions: run: + `ulimit -n 4096 && cargo nextest run -p fabro-acp --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-sandbox --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-workflow --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-validate --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-store --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-server --run-ignored all --no-fail-fast`; + `ulimit -n 4096 && cargo nextest run -p fabro-cli --run-ignored all --no-fail-fast`. + - Expected outcome: every suite passes without skipped tests or live provider credentials. Source of truth: accepted strategy final verification and implementation plan Task 10. + - Interactions: all changed crates and user-visible workflow/server surfaces. + +30. **Workspace-wide build, formatting, and lint gates pass** + - Type: invariant + - Disposition: existing + - Harness: repository-wide Cargo/rustfmt/clippy commands. + - Preconditions: targeted tests pass. + - Actions: run `cargo build --workspace`, `ulimit -n 4096 && cargo nextest run --workspace --run-ignored all --no-fail-fast`, `cargo +nightly-2026-04-14 fmt --check --all`, and `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`. + - Expected outcome: build, workspace tests, formatting, and clippy all pass with zero skipped tests. Source of truth: repository `AGENTS.md` build/test commands and the no-skipped-tests final-run requirement. + - Interactions: full workspace dependency graph, feature flags, generated code boundaries. + +## Coverage Summary + +Covered action space: + +- Workflow authoring: `backend` absent, `api`, `cli`, `acp`, invalid values, stylesheet/import propagation, `acp_command` shell-word and JSON stdio overrides. +- Execution surfaces: agent nodes, prompt/one_shot nodes, local sandbox ACP execution, default command selection, explicit override execution, credentials/env, Node bootstrap, changed-file reporting, cancellation, timeout, and stop reason handling. +- Protocol behavior: `initialize`, `session/new`, `session/prompt`, `session/update` text aggregation, permission requests, `session/cancel`, malformed JSON-RPC, and early process exit. +- Provider/sandbox boundaries: local stdio, Docker non-PTY stdio option/control behavior, Daytona unsupported error, decorator forwarding. +- Product-visible state: ACP events, run projection `provider_used.mode == "acp"`, terminal output/termination, fork replay, CLI workflow run state, and server steerability API behavior. +- Regression protection: existing CLI routing/CLI parsing tests, existing MCP stdio tests, dry-run initialization, repository build/fmt/clippy. + +Explicit exclusions: + +- Live Anthropic/OpenAI/Gemini ACP adapter calls are excluded; fake ACP agents provide deterministic coverage without paid credentials. Risk: vendor-specific adapter quirks may escape until optional/live tests are added. +- Full live Docker ACP workflow execution is not required unless the existing test environment already provides Docker. Unit-level Docker exec option/control tests cover the non-PTY and termination contract. Risk: daemon-specific stream behavior could still differ from Bollard option construction. +- Remote ACP transports are excluded because the implementation plan supports only stdio in this cutover. Risk: none for the agreed scope. +- ACP client filesystem and terminal capabilities are excluded because Fabro intentionally advertises none in this cutover. Risk: agents requiring those client APIs will fail as documented rather than silently using unsafe host capabilities. diff --git a/docs/plans/2026-05-11-add-acp-backend.md b/docs/plans/2026-05-11-add-acp-backend.md new file mode 100644 index 000000000..72d6a77fb --- /dev/null +++ b/docs/plans/2026-05-11-add-acp-backend.md @@ -0,0 +1,1287 @@ +# ACP Backend Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use trycycle-executing to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add `backend="acp"` as a first-class Fabro LLM backend for agent and prompt nodes, backed by the official ACP Rust SDK and isolated in a new `fabro-acp` crate. + +**Architecture:** Add a new `fabro-acp` crate that owns ACP command resolution, sandbox-backed stdio transport, protocol execution, response aggregation, and ACP-specific tests. Extend the sandbox abstraction with bidirectional non-PTY stdio so ACP agents run inside the same local/Docker sandbox where they can read and modify the workspace; `fabro-workflow` keeps the workflow-owned `CodergenBackend` adapter, credentials/env preparation, events, and changed-file detection to avoid leaking workflow concerns into the protocol crate. Change the `CodergenBackend::one_shot` contract to receive the active sandbox and cancel token, because prompt nodes must run ACP inside the workflow sandbox just like agent nodes. Add ACP-specific events and router support so `api`, `cli`, and `acp` are explicit backend choices with no silent fallback for misspellings. + +**Tech Stack:** Rust, Tokio, `agent-client-protocol = "0.11.1"`, `agent-client-protocol-tokio = "0.11.1"` for command parsing/default ACP agent metadata where useful, `agent_client_protocol::Lines` / `ByteStreams` over sandbox stdio, Fabro sandbox providers, `cargo nextest`. + +--- + +## User-Visible Behavior + +- `backend="acp"` on `agent` / `agent_loop` nodes runs an ACP agent turn instead of Fabro's API agent or legacy CLI subprocess parser. +- `backend="acp"` on `prompt` / `one_shot` nodes also runs an ACP prompt turn. ACP cannot universally enforce "no tools" across third-party agents, so the user-visible contract is "ACP prompt turn with text aggregation", not "provider API one-shot with tool use disabled". +- `backend="api"` uses the API backend. A missing `backend` keeps the existing router behavior: API by default, with the current `is_cli_only_model(...)` escape hatch if that list is ever populated. +- `backend="cli"` continues to use the existing CLI backend unchanged except for shared helper extraction. In particular, prompt nodes with `backend="cli"` keep today's API one-shot fallback for compatibility; the router test suite must document that behavior so it is no longer accidental. +- Any other `backend` value fails validation with a clear `unsupported LLM backend` error. Silent fallback to API for misspellings is too risky now that backend choice has behavioral and isolation consequences. +- Default ACP command mapping mirrors current CLI provider mapping: + - Anthropic: `npx -y @zed-industries/claude-code-acp@latest` + - OpenAI, Kimi, Zai, Minimax, Inception, OpenAI-compatible: `npx -y @zed-industries/codex-acp@latest` + - Gemini: `npx -y -- @google/gemini-cli@latest --experimental-acp` +- Before running one of Fabro's default `npx`-based ACP commands, Fabro ensures Node/npm/npx exist in the sandbox using the same Node bootstrap strategy as the legacy CLI backend. Explicit `acp_command` overrides are not implicitly installed beyond this Node bootstrap; if they need other binaries, the workflow/sandbox image must provide them. +- Advanced users and tests can set `acp_command="..."` on an ACP-backed node to override the default command. The override is only honored when `backend="acp"` and is parsed with `agent_client_protocol_tokio::AcpAgent::from_str(...)`; Fabro supports only parsed `McpServer::Stdio` commands in this cutover. JSON stdio server configs are valid, HTTP/SSE configs are rejected with a clear unsupported-command error, and parsed program/args/env are re-rendered for sandbox execution with `shell_quote()` instead of executing the raw attribute string through the shell. +- ACP receives the same provider credentials and workflow tool env currently forwarded to CLI agents. Model selection is recorded in Fabro events/projections, but stable ACP v1 has no portable model-selection request. Users who need model-specific ACP behavior must encode that in their chosen ACP command until ACP model/session config stabilizes. +- ACP stages emit `agent.acp.started`, `agent.acp.completed`, `agent.acp.cancelled`, and `agent.acp.timed_out`; run projections expose `provider_used.mode == "acp"`. +- ACP support is implemented for local and Docker sandboxes in this cutover. Daytona gets an explicit unsupported-provider error for ACP because the current Daytona command API does not expose raw bidirectional stdio; legacy `backend="cli"` remains available there. This is deliberate: running ACP on the host or over a PTY would violate isolation or corrupt JSON-RPC framing. +- Fabro handles ACP `session/request_permission` requests by auto-selecting an allow option, matching the trust model of existing CLI mode flags such as `--full-auto`, `--yolo`, and `--dangerously-skip-permissions`. Prefer an `AllowAlways` option when present, then `AllowOnce`, then the first non-reject option; if no allow option exists, return `RequestPermissionOutcome::Cancelled`. +- Fabro advertises no ACP client filesystem or terminal capabilities in this cutover. ACP agents that need those client-side APIs may fail with method-not-found; the supported path is ACP adapters that operate through their own process in the sandbox, which matches Fabro's existing CLI-agent isolation model. + +## Contracts And Invariants + +- ACP agent processes must run inside the active Fabro sandbox, not on the host, so file mutations, Git diff detection, secrets forwarding, and cancellation match existing run isolation. +- ACP stdio must be line-preserving, non-PTY JSON-RPC. Do not implement ACP over terminal PTY sessions. +- Do not use `agent_client_protocol_tokio::AcpAgent` to connect to the running agent process; it spawns on the host. It is acceptable only for parsing/validating command strings or using its default command constants. The actual connection must adapt `Sandbox::spawn_stdio_process(...)` into an `agent_client_protocol` transport. +- Do not accept an `acp_command` by validating it with the ACP Tokio helper and then executing the original raw string. Store the parsed stdio server command, args, env, and display string; use the parsed representation for execution so JSON stdio configs and shell-word overrides behave consistently. +- The new `fabro-acp` crate must use `agent-client-protocol` schema/session/message types for initialization, session creation, prompt turns, updates, cancellation, and fake-agent tests. Do not hand-roll ACP request/response structs. +- `fabro-acp` must not depend on `fabro-workflow`; otherwise `fabro-workflow` cannot instantiate it without a dependency cycle. The workflow adapter is intentionally thin and delegates all protocol behavior to `fabro-acp`. +- ACP response text is the concatenation of `SessionUpdate::AgentMessageChunk(ContentChunk { content: ContentBlock::Text(...), .. })` chunks until the prompt response stop reason arrives. Thought chunks, plans, tool call updates, and custom updates are ignored for `CodergenResult::Text` but must keep the stall watchdog alive. +- ACP permission requests must be handled through the official `RequestPermissionRequest` / `RequestPermissionResponse` schema types. Do not ignore them: real coding-agent adapters may request permission before file edits even when filesystem and terminal client capabilities are not advertised. +- Stop reason handling: + - `EndTurn` and `Refusal`: return text as the stage response. + - `Cancelled`: emit `agent.acp.cancelled` and return `Error::Cancelled`. + - `MaxTokens` / `MaxTurnRequests`: emit completion with the partial output and return a handler error containing the stop reason. +- Cancellation must attempt ACP `session/cancel` when a session exists, then terminate the stdio process if the agent does not finish promptly. +- Timeouts use `node.timeout()` like CLI mode. A timeout emits `agent.acp.timed_out`, terminates the ACP process, and returns a handler error. +- Files touched are detected by comparing sandbox Git state before and after the ACP turn, using the same semantics as CLI mode: changed tracked files plus untracked files, sorted and deduplicated, then filtered against pre-existing dirty files. + +## File Structure + +- Create `lib/crates/fabro-acp/Cargo.toml` and `lib/crates/fabro-acp/src/lib.rs`: crate surface and exports. +- Create `lib/crates/fabro-acp/src/command.rs`: provider-to-ACP-command mapping and command override parsing helpers. +- Create `lib/crates/fabro-acp/src/transport.rs`: adapt `fabro_sandbox::Sandbox::spawn_stdio_process(...)` into an `agent_client_protocol` transport using `Lines` or `ByteStreams`, collect stderr tails, and terminate/wait on process cleanup. +- Create `lib/crates/fabro-acp/src/session.rs`: ACP lifecycle using `agent_client_protocol::Client`, `InitializeRequest`, `NewSessionRequest`, `PromptRequest`, `SessionUpdate`, `CancelNotification`, and stop reason handling. +- Create `lib/crates/fabro-acp/src/error.rs`: ACP-specific error type that converts cleanly into workflow handler errors. +- Create `lib/crates/fabro-acp/src/test_support.rs` behind `#[cfg(any(test, feature = "test-support"))]`: fake ACP agent/transport helpers using `agent-client-protocol` types. +- Modify root `Cargo.toml`: add workspace dependencies for `agent-client-protocol` and `agent-client-protocol-tokio` and include `fabro-acp` through the existing `lib/crates/*` workspace glob. +- Modify `lib/crates/fabro-agent/src/sandbox.rs`, `lib/crates/fabro-agent/src/lib.rs`, and `lib/crates/fabro-sandbox/src/sandbox.rs`: expose the new sandbox stdio process API and public process/handle/stderr-tail types through the existing sandbox API re-export path so callers using either `fabro_sandbox::*` or `fabro_agent::sandbox::*` can compile without reaching into private modules. +- Modify `lib/crates/fabro-sandbox/src/local.rs`: implement bidirectional stdio process spawning. +- Modify `lib/crates/fabro-sandbox/src/docker.rs`: implement bidirectional Docker exec stdio without TTY. +- Modify `lib/crates/fabro-sandbox/src/daytona/mod.rs`: return an explicit unsupported error for bidirectional stdio. +- Modify `lib/crates/fabro-sandbox/src/worktree.rs`, `read_guard.rs`, and sandbox decorators: forward stdio support to wrapped sandboxes and preserve worktree path resolution. +- Create `lib/crates/fabro-workflow/src/handler/llm/acp.rs`: workflow-owned `CodergenBackend` adapter that calls `fabro_acp`. +- Create `lib/crates/fabro-workflow/src/handler/llm/changed_files.rs`: shared Git changed-file detection currently embedded in CLI backend. +- Create `lib/crates/fabro-workflow/src/handler/llm/node_runtime.rs`: shared Node/npm bootstrap helper used by both CLI and ACP default `npx` commands. +- Modify `lib/crates/fabro-workflow/src/handler/llm/cli.rs`: use shared changed-file helpers and move `BackendRouter` to support API/CLI/ACP selection. +- Modify `lib/crates/fabro-workflow/src/handler/llm/mod.rs`: export `AgentAcpBackend` and the router. +- Modify `lib/crates/fabro-workflow/src/handler/agent.rs`, `prompt.rs`, `llm/api.rs`, tests, and any `CodergenBackend` stubs: extend `one_shot` with `sandbox` and `cancel_token` parameters so ACP prompt nodes can run in the active sandbox. +- Modify `lib/crates/fabro-types/src/graph.rs`: add `Node::acp_command()`. +- Modify `lib/crates/fabro-workflow/src/transforms/import.rs`: treat `acp_command` as a semantic default attribute when imported workflow placeholders carry it. +- Create `lib/crates/fabro-validate/src/rules/backend_valid.rs` and modify `lib/crates/fabro-validate/src/rules/mod.rs`: validate node `backend` values are absent or one of `api`, `cli`, `acp`. +- Modify `lib/crates/fabro-workflow/src/pipeline/initialize.rs`: construct `AgentAcpBackend` alongside API and CLI backends. +- Modify event files: `lib/crates/fabro-workflow/src/event/events.rs`, `names.rs`, `convert.rs`, and `stored_fields.rs` for `agent.acp.*`. +- Modify fork replay: `lib/crates/fabro-workflow/src/operations/fork.rs` so ACP provider metadata and terminal status survive fork projection rebuilds. +- Modify run event/projection files: `lib/crates/fabro-types/src/run_event/mod.rs`, `lib/crates/fabro-types/src/run_event/misc.rs`, and `lib/crates/fabro-store/src/run_state.rs`. +- Modify server steerability tracking: `lib/crates/fabro-server/src/server.rs`, `lib/crates/fabro-server/src/server/handler/steer.rs`, and `lib/crates/fabro-server/src/server/tests.rs` so currently running ACP stages are treated as non-steerable like CLI-backed stages instead of allowing buffered API steering. +- Modify docs: `docs/public/reference/dot-language.mdx`, `docs/public/core-concepts/agents.mdx`, and any CLI/backend reference that currently says only `api`/`cli`. +- Add/update tests in `lib/crates/fabro-acp/tests/`, `lib/crates/fabro-sandbox` unit tests, `lib/crates/fabro-workflow/tests/it/integration.rs`, `lib/crates/fabro-store/src/run_state.rs`, `lib/crates/fabro-server/src/server/tests.rs`, and `lib/crates/fabro-cli/tests/it/workflow/`. + +## Task 1: Read Strategy Docs And Pin Protocol API + +**Files:** +- Read: `docs/internal/events-strategy.md` +- Read: `docs/internal/testing-strategy.md` +- Read: `docs/internal/error-handling-strategy.md` +- Read: `https://docs.rs/agent-client-protocol/latest/agent_client_protocol/` +- Read: `https://docs.rs/agent-client-protocol-tokio/latest/agent_client_protocol_tokio/` + +- [ ] **Step 1: Confirm repo strategy constraints** + +Read the three internal strategy docs before changing events, tests, or error paths. Capture any additional constraints in short implementation notes inside the task branch, not in committed docs unless the implementation needs them. + +- [ ] **Step 2: Confirm ACP SDK API locally** + +Run: + +```bash +cargo info agent-client-protocol +cargo info agent-client-protocol-tokio +``` + +Expected: current latest is `0.11.1` for both crates. If newer versions are available, use the latest compatible version and update this plan's exact version references in the implementation commit message. + +- [ ] **Step 3: Inspect ACP crate examples/source** + +Run: + +```bash +rg -n "build_session|send_prompt|read_update|SessionUpdate|AcpAgent|zed_codex|google_gemini" ~/.cargo/registry/src -g '*.rs' +``` + +Expected: identify the SDK's `Client.builder()`, `InitializeRequest::new(ProtocolVersion::V1)`, `ConnectionTo::build_session`, `ActiveSession::send_prompt`, and `SessionUpdate` types. + +- [ ] **Step 4: Commit** + +No code changes are expected in this task. + +## Task 2: Add `fabro-acp` Crate Skeleton And Command Mapping + +**Files:** +- Create: `lib/crates/fabro-acp/Cargo.toml` +- Create: `lib/crates/fabro-acp/src/lib.rs` +- Create: `lib/crates/fabro-acp/src/command.rs` +- Test: `lib/crates/fabro-acp/src/command.rs` +- Modify: `Cargo.toml` + +- [ ] **Step 1: Write failing command mapping tests** + +Add tests proving provider mapping: + +```rust +#[test] +fn default_command_for_anthropic_uses_zed_claude_acp() { + assert_eq!( + default_acp_command(Provider::Anthropic).to_string(), + "npx -y @zed-industries/claude-code-acp@latest" + ); +} + +#[test] +fn default_command_for_openai_compatible_family_uses_zed_codex_acp() { + for provider in [ + Provider::OpenAi, + Provider::Kimi, + Provider::Zai, + Provider::Minimax, + Provider::Inception, + Provider::OpenAiCompatible, + ] { + assert_eq!( + default_acp_command(provider).to_string(), + "npx -y @zed-industries/codex-acp@latest" + ); + } +} + +#[test] +fn default_command_for_gemini_uses_experimental_acp() { + assert_eq!( + default_acp_command(Provider::Gemini).to_string(), + "npx -y -- @google/gemini-cli@latest --experimental-acp" + ); +} +``` + +Add override parsing tests: + +```rust +#[test] +fn explicit_acp_command_overrides_provider_default() { + let command = resolve_acp_command(Provider::OpenAi, Some("python fake_agent.py")).unwrap(); + assert_eq!(command.to_string(), "python fake_agent.py"); + assert_eq!(command.program(), Path::new("python")); + assert_eq!(command.args(), &["fake_agent.py".to_string()]); +} + +#[test] +fn blank_acp_command_is_rejected() { + let err = resolve_acp_command(Provider::OpenAi, Some(" ")).unwrap_err(); + assert!(err.to_string().contains("acp_command must not be empty")); +} + +#[test] +fn json_stdio_acp_command_is_supported() { + let raw = r#"{"type":"stdio","name":"fake","command":"python","args":["fake agent.py"],"env":[{"name":"MODE","value":"test"}]}"#; + let command = resolve_acp_command(Provider::OpenAi, Some(raw)).unwrap(); + assert_eq!(command.program(), Path::new("python")); + assert_eq!(command.args(), &["fake agent.py".to_string()]); + assert_eq!(command.env().get("MODE").map(String::as_str), Some("test")); +} + +#[test] +fn non_stdio_acp_command_is_rejected() { + let raw = r#"{"type":"http","name":"remote","url":"https://example.test/acp"}"#; + let err = resolve_acp_command(Provider::OpenAi, Some(raw)).unwrap_err(); + assert!(err.to_string().contains("only stdio ACP commands are supported")); +} +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-acp command +``` + +Expected: FAIL because `fabro-acp` and `default_acp_command` do not exist. + +- [ ] **Step 3: Implement crate skeleton** + +Add dependencies: + +```toml +[features] +test-support = [] + +[dependencies] +agent-client-protocol.workspace = true +agent-client-protocol-tokio.workspace = true +fabro-model = { path = "../fabro-model" } +fabro-sandbox = { path = "../fabro-sandbox" } +fabro-types = { path = "../fabro-types" } +fabro-util = { path = "../fabro-util" } +bytes.workspace = true +serde.workspace = true +serde_json.workspace = true +thiserror.workspace = true +tokio.workspace = true +tokio-util = { workspace = true, features = ["compat", "io"] } +futures.workspace = true +uuid.workspace = true +tracing.workspace = true + +[dev-dependencies] +tempfile = "3" +``` + +Add workspace dependencies in root `Cargo.toml`: + +```toml +agent-client-protocol = "0.11.1" +agent-client-protocol-tokio = "0.11.1" +``` + +Implement command resolution around the ACP SDK's parsed stdio server representation, not a raw shell string: + +```rust +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AcpCommand { + display: String, + program: PathBuf, + args: Vec, + env: HashMap, +} + +impl AcpCommand { + pub fn program(&self) -> &Path { &self.program } + pub fn args(&self) -> &[String] { &self.args } + pub fn env(&self) -> &HashMap { &self.env } + pub fn display(&self) -> &str { &self.display } + + pub fn to_shell_command(&self) -> String { + std::iter::once(self.program.to_string_lossy().into_owned()) + .chain(self.args.iter().cloned()) + .map(|part| fabro_sandbox::shell_quote(&part)) + .collect::>() + .join(" ") + } +} + +impl std::fmt::Display for AcpCommand { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.display) + } +} + +pub fn default_acp_command(provider: fabro_model::Provider) -> AcpCommand { + match provider { + fabro_model::Provider::Anthropic => { + parse_acp_command("npx -y @zed-industries/claude-code-acp@latest").unwrap() + } + fabro_model::Provider::Gemini => { + parse_acp_command("npx -y -- @google/gemini-cli@latest --experimental-acp").unwrap() + } + fabro_model::Provider::OpenAi + | fabro_model::Provider::Kimi + | fabro_model::Provider::Zai + | fabro_model::Provider::Minimax + | fabro_model::Provider::Inception + | fabro_model::Provider::OpenAiCompatible => { + parse_acp_command("npx -y @zed-industries/codex-acp@latest").unwrap() + } + } +} + +pub fn resolve_acp_command( + provider: fabro_model::Provider, + override_command: Option<&str>, +) -> Result { + if let Some(raw) = override_command { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return Err(AcpCommandError::EmptyOverride); + } + return parse_acp_command(trimmed); + } + Ok(default_acp_command(provider)) +} + +fn parse_acp_command(raw: &str) -> Result { + let agent = agent_client_protocol_tokio::AcpAgent::from_str(raw)?; + let agent_client_protocol::schema::McpServer::Stdio(stdio) = agent.into_server() else { + return Err(AcpCommandError::UnsupportedTransport); + }; + Ok(AcpCommand { + display: raw.to_string(), + program: stdio.command, + args: stdio.args, + env: stdio.env.into_iter().map(|env| (env.name, env.value)).collect(), + }) +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-acp command +``` + +Expected: PASS. + +- [ ] **Step 5: Refactor and verify** + +Keep the crate API narrow: export `AcpCommand`, `default_acp_command`, and no workflow types. + +Run: + +```bash +cargo build -p fabro-acp +``` + +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add Cargo.toml lib/crates/fabro-acp +git commit -m "feat: add ACP backend crate skeleton" +``` + +## Task 3: Add Sandbox Bidirectional Stdio Capability + +**Files:** +- Modify: `lib/crates/fabro-sandbox/src/sandbox.rs` +- Modify: `lib/crates/fabro-sandbox/src/local.rs` +- Modify: `lib/crates/fabro-sandbox/src/docker.rs` +- Modify: `lib/crates/fabro-sandbox/src/daytona/mod.rs` +- Modify: `lib/crates/fabro-sandbox/src/worktree.rs` +- Modify: `lib/crates/fabro-sandbox/src/read_guard.rs` +- Modify: `lib/crates/fabro-sandbox/src/sandbox.rs` `delegate_sandbox!` macro +- Modify: `lib/crates/fabro-sandbox/src/test_support.rs` +- Modify: `lib/crates/fabro-agent/src/lib.rs` +- Test: provider-local unit tests in `lib/crates/fabro-sandbox/src/local.rs` +- Test: Docker option/unit tests in `lib/crates/fabro-sandbox/src/docker.rs` +- Test: decorator forwarding tests in `lib/crates/fabro-sandbox/src/read_guard.rs` or `test_support.rs` + +- [ ] **Step 1: Write failing local stdio test** + +Add a local sandbox test that starts a line-oriented process and round-trips stdin/stdout: + +```rust +#[tokio::test] +async fn stdio_process_round_trips_lines() { + let tempdir = tempfile::tempdir().unwrap(); + let sandbox = LocalSandbox::new(tempdir.path().to_path_buf()); + let mut process = sandbox + .spawn_stdio_process( + "python3 -u -c 'import sys; [print(line.strip()[::-1], flush=True) for line in sys.stdin]'", + None, + None, + None, + ) + .await + .unwrap(); + + process.write_line("abc").await.unwrap(); + assert_eq!(process.read_stdout_line().await.unwrap(), Some("cba".to_string())); + process.terminate().await.unwrap(); +} +``` + +The exact helper names can differ, but the test must prove bidirectional stdio, not just command output streaming. + +- [ ] **Step 2: Run test to verify it fails** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-sandbox stdio_process_round_trips_lines +``` + +Expected: FAIL because the sandbox stdio API does not exist. + +- [ ] **Step 3: Implement API and local provider** + +Add an object-safe sandbox method with a default unsupported implementation: + +```rust +async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&HashMap>, + cancel_token: Option, +) -> crate::Result; +``` + +`StdioProcess` should own stdin, stdout, stderr, and child lifecycle. It must expose enough typed IO for `fabro-acp` to build an `agent_client_protocol` transport without provider-specific downcasts: + +```rust +pub struct StdioProcess { + pub stdin: Pin>, + pub stdout: Pin>, + pub stderr: StderrCollector, + pub handle: StdioProcessHandle, +} + +impl StdioProcessHandle { + pub async fn terminate(&self) -> crate::Result<()>; + pub async fn wait(&self) -> crate::Result; +} +``` + +The exact type names can differ, but the public API must support all of these operations: + +- build line-oriented JSON-RPC from stdout/stdin in `fabro-acp` +- collect stderr concurrently for protocol/exit errors +- terminate the child/exec on timeout or cancellation +- wait for final termination without leaking provider internals + +For local, spawn `/bin/bash -lc ` with piped stdin/stdout/stderr, current env filtering consistent with `exec_command_streaming`, and process-group cleanup. The local implementation can expose child stdout directly as `AsyncRead`. + +- [ ] **Step 4: Implement Docker provider** + +Use Docker exec with non-PTY stdio and explicit cancellation support. Do not rely on Docker having an exec-kill API. Adapt the existing `docker_controlled_shell_command(...)` / stop-file strategy used by `exec_command_streaming` so `StdioProcessHandle::terminate()` can signal the wrapper from a separate exec, wait briefly, and then force-kill the recorded process group when needed. + +Create exec with: + +```rust +CreateExecOptions { + attach_stdin: Some(true), + attach_stdout: Some(true), + attach_stderr: Some(true), + tty: Some(false), + cmd: Some(vec!["/bin/bash".to_string(), "-lc".to_string(), command.to_string()]), + working_dir: Some(effective_dir), + env: Some(env_vec), + ..Default::default() +} +``` + +Start with `StartExecOptions { detach: false, tty: false, output_capacity: None }` and keep the returned `input` writer. Bollard returns stdout/stderr as a multiplexed `Stream` rather than separate `AsyncRead`s; convert only `LogOutput::StdOut` bytes into the `stdout` reader used by ACP and feed `LogOutput::StdErr` bytes into the stderr collector. The test must assert both create and start options use `tty == false` because ACP JSON-RPC must not run over PTY. + +Add Docker unit tests around the option-builder/control wrapper proving: + +- `attach_stdin`, `attach_stdout`, and `attach_stderr` are true +- create/start `tty` is false +- the controlled command writes a pid file and reacts to the stop file +- `terminate()` uses the stop-file path rather than silently dropping the stream + +- [ ] **Step 5: Implement provider forwarding and unsupported Daytona** + +Forward through `WorktreeSandbox`, read/write decorators, test-support sandboxes, and the `delegate_sandbox!` macro. Re-export the new public stdio process types from both `fabro-sandbox/src/lib.rs` and `fabro-agent/src/sandbox.rs` / `fabro-agent/src/lib.rs` alongside the existing sandbox exports. A decorator must not inherit the default unsupported implementation when its inner sandbox supports stdio. Daytona should return an unsupported error like: + +```text +ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet +``` + +- [ ] **Step 6: Run tests to verify they pass** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-sandbox stdio +``` + +Expected: PASS for local/unit stdio tests; no live Docker requirement unless an existing Docker test harness is available. + +- [ ] **Step 7: Refactor and verify** + +Keep the stdio API provider-neutral. Do not leak Bollard or Tokio child types through public trait signatures. + +Run: + +```bash +cargo build -p fabro-sandbox +``` + +Expected: PASS. + +- [ ] **Step 8: Commit** + +```bash +git add lib/crates/fabro-sandbox lib/crates/fabro-agent/src/lib.rs lib/crates/fabro-agent/src/sandbox.rs +git commit -m "feat: add sandbox stdio processes" +``` + +## Task 4: Implement ACP Session Lifecycle In `fabro-acp` + +**Files:** +- Create: `lib/crates/fabro-acp/src/session.rs` +- Create: `lib/crates/fabro-acp/src/transport.rs` +- Create: `lib/crates/fabro-acp/src/error.rs` +- Create: `lib/crates/fabro-acp/src/test_support.rs` +- Modify: `lib/crates/fabro-acp/src/lib.rs` +- Test: `lib/crates/fabro-acp/tests/session.rs` + +- [ ] **Step 1: Write fake-agent lifecycle test** + +Use `agent-client-protocol` request/response/schema types in the fake agent. The test must assert the observed method order: + +```text +initialize +session/new +session/prompt +``` + +It must also assert that `SessionUpdate::AgentMessageChunk(ContentChunk { content: ContentBlock::Text(...), .. })` chunks are concatenated into the returned text. + +- [ ] **Step 2: Run test to verify it fails** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-acp session_lifecycle +``` + +Expected: FAIL because `run_acp_turn` does not exist. + +- [ ] **Step 3: Implement `AcpRunRequest` / `AcpRunResult`** + +Use an API that is neutral to `fabro-workflow` and depends directly on `fabro-sandbox` for the sandbox trait: + +```rust +pub struct AcpRunRequest { + pub command: AcpCommand, + pub prompt: String, + pub cwd: String, + pub timeout_ms: Option, + pub env: HashMap, + pub sandbox: Arc, + pub cancel_token: CancellationToken, + pub on_activity: Option>, +} + +pub struct AcpRunResult { + pub text: String, + pub stop_reason: agent_client_protocol::schema::StopReason, + pub stderr: String, + pub duration_ms: u64, +} +``` + +Keep usage optional/absent for now because stable ACP v1 does not provide portable token usage without unstable features. + +- [ ] **Step 4: Implement sandbox-backed ACP transport** + +Implement a `SandboxAcpTransport` in `transport.rs` that implements `agent_client_protocol::ConnectTo` by: + +- building the launch environment from `request.command.env()` first, then overlaying `request.env` from the workflow adapter so Fabro-managed credentials, login results, and tool env win over duplicate keys supplied in an `acp_command` JSON config. This matches the legacy CLI backend's launch-env behavior and prevents a command override from accidentally shadowing refreshed credentials. +- calling `request.sandbox.spawn_stdio_process(request.command.to_shell_command(), ...)` +- adapting process stdout/stdin into `agent_client_protocol::Lines` or `agent_client_protocol::ByteStreams` +- collecting stderr concurrently into a bounded tail for errors +- racing protocol completion against early process exit +- terminating and waiting for the process on timeout/cancellation/error + +Use `tokio_util::compat::{TokioAsyncReadCompatExt, TokioAsyncWriteCompatExt}` when converting Tokio IO to the `futures` IO traits used by `agent-client-protocol`. Do not use `agent_client_protocol_tokio::AcpAgent` for this connection because it launches the command on the host. + +- [ ] **Step 5: Implement ACP lifecycle** + +Use the official SDK and register a client-side permission handler before connecting: + +```rust +use agent_client_protocol::schema::{ + InitializeRequest, PermissionOptionKind, ProtocolVersion, RequestPermissionOutcome, + RequestPermissionRequest, RequestPermissionResponse, SelectedPermissionOutcome, +}; + +let permission_cancel_token = request.cancel_token.clone(); +Client.builder() + .name("fabro") + .on_receive_request( + async move |request: RequestPermissionRequest, responder, _connection| { + if permission_cancel_token.is_cancelled() { + return responder.respond(RequestPermissionResponse::new( + RequestPermissionOutcome::Cancelled, + )); + } + let selected = request + .options + .iter() + .find(|option| option.kind == PermissionOptionKind::AllowAlways) + .or_else(|| { + request + .options + .iter() + .find(|option| option.kind == PermissionOptionKind::AllowOnce) + }) + .or_else(|| { + request.options.iter().find(|option| { + !matches!( + option.kind, + PermissionOptionKind::RejectOnce | PermissionOptionKind::RejectAlways + ) + }) + }); + let outcome = selected.map_or(RequestPermissionOutcome::Cancelled, |option| { + RequestPermissionOutcome::Selected(SelectedPermissionOutcome::new( + option.option_id.clone(), + )) + }); + responder.respond(RequestPermissionResponse::new(outcome)) + }, + agent_client_protocol::on_receive_request!(), + ) + .connect_with(SandboxAcpTransport::new(&request), async |cx| { + cx.send_request(InitializeRequest::new(ProtocolVersion::V1)) + .block_task() + .await?; + cx.build_session(&request.cwd) + .block_task() + .run_until(async |mut session| { + session.send_prompt(request.prompt)?; + read_turn(&mut session).await + }) + .await + }) + .await +``` + +Use lower-level `read_update()` rather than only `read_to_string()` so the implementation can capture stop reasons, call `on_activity` for every update/response, and handle non-text updates deterministically. + +When reading updates, handle cancellation with `tokio::select!`: + +- if the cancel token fires after a session exists, send `CancelNotification::new(session_id.clone())` with `session.connection().send_notification_to(Agent, ...)` +- keep draining until the agent returns `StopReason::Cancelled`, or terminate the process after a short grace period +- if cancellation fires before a session exists, terminate the process and return `AcpError::Cancelled` + +The initialize request should use `InitializeRequest::new(ProtocolVersion::V1)` and the default empty `ClientCapabilities`; do not advertise filesystem or terminal support until handlers for those requests are implemented. + +Permission approval is intentionally automatic in this cutover because Fabro's legacy CLI backend already runs coding CLIs with all tool approvals bypassed. If cancellation has already fired when a permission request arrives, respond with `RequestPermissionOutcome::Cancelled`. + +- [ ] **Step 6: Add cancellation and timeout tests** + +Tests must cover: + +- permission requests select an allow option and let the ACP turn continue +- permission requests after cancellation respond with `RequestPermissionOutcome::Cancelled` +- cancellation before prompt completion sends `CancelNotification::new(session_id)` / `session/cancel` when a session exists and returns `AcpError::Cancelled` +- timeout terminates the stdio process and returns `AcpError::TimedOut` +- malformed JSON-RPC from the agent returns a protocol error with stderr tail if present +- early process exit returns an error that includes exit status/stderr + +- [ ] **Step 7: Run ACP crate tests** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-acp +``` + +Expected: PASS. + +- [ ] **Step 8: Refactor and verify** + +Ensure all public error messages are suitable for surfacing as handler errors. Avoid exposing test helpers outside `#[cfg(any(test, feature = "test-support"))]`. + +Run: + +```bash +cargo build -p fabro-acp +``` + +Expected: PASS. + +- [ ] **Step 9: Commit** + +```bash +git add lib/crates/fabro-acp +git commit -m "feat: implement ACP session client" +``` + +## Task 5: Add Workflow ACP Backend Adapter And Router Support + +**Files:** +- Create: `lib/crates/fabro-workflow/src/handler/llm/acp.rs` +- Create: `lib/crates/fabro-workflow/src/handler/llm/changed_files.rs` +- Create: `lib/crates/fabro-workflow/src/handler/llm/node_runtime.rs` +- Modify: `lib/crates/fabro-workflow/src/handler/agent.rs` +- Modify: `lib/crates/fabro-workflow/src/handler/prompt.rs` +- Modify: `lib/crates/fabro-workflow/src/handler/llm/api.rs` +- Modify: `lib/crates/fabro-workflow/src/handler/llm/cli.rs` +- Modify: `lib/crates/fabro-workflow/src/handler/llm/mod.rs` +- Modify: `lib/crates/fabro-workflow/Cargo.toml` +- Modify: `lib/crates/fabro-types/src/graph.rs` +- Modify: `lib/crates/fabro-workflow/src/transforms/import.rs` +- Create: `lib/crates/fabro-validate/src/rules/backend_valid.rs` +- Modify: `lib/crates/fabro-validate/src/rules/mod.rs` +- Test: `lib/crates/fabro-workflow/src/handler/llm/acp.rs` +- Test: `lib/crates/fabro-workflow/src/handler/llm/cli.rs` +- Test: `lib/crates/fabro-validate/src/rules/backend_valid.rs` + +- [ ] **Step 1: Write failing backend adapter tests** + +Add tests that prove: + +- `AgentAcpBackend::run` sends the node prompt to `fabro-acp` and returns `CodergenResult::Text` +- `AgentAcpBackend::run` honors node `acp_command` only when routing to ACP +- `PromptHandler` passes the active sandbox and run cancel token through `CodergenBackend::one_shot` +- `AgentAcpBackend::one_shot` combines `system_prompt` and `prompt` into a single ACP prompt and runs it through the passed sandbox, not the host +- default ACP commands bootstrap Node/npx before launch when Node is absent +- explicit `acp_command` does not trigger provider CLI installation and is still run inside the sandbox +- stop reason `Cancelled` maps to `Error::Cancelled` +- max-token/max-turn stop reasons map to handler errors +- files touched are computed relative to pre-existing dirty files + +- [ ] **Step 2: Write failing router tests** + +Update router tests to cover: + +```rust +router_uses_api_by_default +router_uses_api_for_backend_api +router_uses_cli_for_backend_cli +router_uses_acp_for_backend_acp +router_rejects_unknown_backend +router_routes_one_shot_to_acp_for_backend_acp +router_routes_one_shot_to_api_by_default +``` + +- [ ] **Step 3: Write failing backend validation tests** + +Add a `backend_valid` rule test proving `backend="api"`, `backend="cli"`, `backend="acp"`, and absent backend are accepted, while `backend="codex"` returns an error diagnostic containing: + +```text +unsupported LLM backend "codex"; expected one of: api, cli, acp +``` + +- [ ] **Step 4: Run tests to verify they fail** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_uses_acp_for_backend_acp) | test(router_routes_one_shot_to_acp_for_backend_acp) | test(acp_backend)' +ulimit -n 4096 && cargo nextest run -p fabro-validate -E 'test(backend_valid)' +``` + +Expected: FAIL because ACP adapter/router support and backend validation do not exist. + +- [ ] **Step 5: Extend `CodergenBackend::one_shot` for sandboxed backends** + +Change the trait method signature in `handler/agent.rs` to include the active sandbox and cancel token: + +```rust +async fn one_shot( + &self, + node: &Node, + prompt: &str, + system_prompt: Option<&str>, + emitter: &Arc, + stage_scope: &StageScope, + sandbox: &Arc, + cancel_token: CancellationToken, +) -> Result +``` + +Update `PromptHandler::execute` to pass `&services.run.sandbox` and `services.run.cancel_token()`. Update `AgentApiBackend`, `BackendRouter`, and all test stubs to accept the new parameters. API-backed one-shot calls should ignore these new parameters; ACP-backed one-shot calls must use them. + +- [ ] **Step 6: Implement shared changed-file helpers** + +Move CLI duplicated logic into `changed_files.rs`: + +```rust +pub async fn detect_changed_files(sandbox: &Arc) -> Vec; +pub async fn files_touched_since( + sandbox: &Arc, + files_before: &[String], +) -> (Vec, Option); +``` + +Use `shell_quote()` for the `ls -t` command. Update CLI backend to call these helpers without behavior changes. + +- [ ] **Step 7: Implement ACP env preparation and Node bootstrap** + +Mirror CLI credential behavior in the workflow adapter before calling `fabro_acp`: + +- If a `CredentialResolver` exists, resolve `CredentialUsage::CliAgent(CliAgentKind::{Claude,Codex,Gemini})`. +- Run any credential `login_command` in the sandbox before starting ACP. +- Forward credential `env_vars`. +- Merge workflow tool env provider values. +- Preserve the GitHub token refresh notice behavior in the workflow adapter, because notices are workflow events. +- Ensure Node/npm/npx exist before running the default `npx` ACP commands. Extract the existing Node installation shell from `ensure_cli` into a shared helper instead of duplicating a second hardcoded tarball command. + +- [ ] **Step 8: Implement `AgentAcpBackend`** + +The adapter owns model/provider/resolver/tool env configuration like `AgentCliBackend`, builds `AcpRunRequest` with the active sandbox, cancellation token, and an `on_activity` callback that calls `Emitter::touch`, then delegates to `fabro_acp`. Defer ACP-specific started/completed/cancelled/timed-out event emission to Task 6, where the event variants and projection support are added in the same commit. + +For `one_shot`, build the ACP prompt as: + +```text +System: +{system_prompt} + +User: +{prompt} +``` + +If `system_prompt` is `None` or empty, send only `{prompt}`. + +- [ ] **Step 9: Implement three-way `BackendRouter`** + +Change router fields to: + +```rust +api_backend: Box, +cli_backend: AgentCliBackend, +acp_backend: AgentAcpBackend, +``` + +Route by parsed backend enum: + +```rust +enum SelectedBackend { Api, Cli, Acp } +``` + +Selection rules: + +- `None` -> CLI only if `is_cli_only_model(model)`, otherwise API +- `"api"` -> API +- `"cli"` -> CLI +- `"acp"` -> ACP +- anything else -> validation error + +For `one_shot`, route `"acp"` to ACP and all other valid values to API. Keep the existing `backend="cli"` prompt-node API fallback for backward compatibility, but add a test documenting that legacy behavior so it is no longer accidental. Do not silently route `backend="acp"` to API. + +- [ ] **Step 10: Implement backend validation** + +Add `backend_valid::rule()` to `fabro-validate` and register it in `built_in_rules()`. This complements router runtime errors and makes `fabro validate` catch misspellings before execution. + +- [ ] **Step 11: Run tests to verify they pass** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_uses_cli_for_backend_attr) | test(router_uses_api_by_default) | test(router_uses_acp_for_backend_acp) | test(router_routes_one_shot_to_acp_for_backend_acp) | test(agent_cli_backend_run_writes_prompt_and_calls_exec) | test(acp_backend)' +ulimit -n 4096 && cargo nextest run -p fabro-validate -E 'test(backend_valid)' +``` + +Expected: PASS. + +- [ ] **Step 12: Refactor and verify** + +Keep ACP-specific protocol code out of `fabro-workflow`; the adapter should translate workflow concepts to `fabro-acp` requests and back. + +Run: + +```bash +cargo build -p fabro-workflow -p fabro-validate +``` + +Expected: PASS. + +- [ ] **Step 13: Commit** + +```bash +git add lib/crates/fabro-workflow lib/crates/fabro-workflow/Cargo.toml lib/crates/fabro-types/src/graph.rs lib/crates/fabro-validate/src/rules +git commit -m "feat: route workflow stages to ACP backend" +``` + +## Task 6: Add ACP Events And Run Projection Support + +**Files:** +- Modify: `lib/crates/fabro-workflow/src/handler/llm/acp.rs` +- Modify: `lib/crates/fabro-workflow/src/event/events.rs` +- Modify: `lib/crates/fabro-workflow/src/event/names.rs` +- Modify: `lib/crates/fabro-workflow/src/event/convert.rs` +- Modify: `lib/crates/fabro-workflow/src/event/stored_fields.rs` +- Modify: `lib/crates/fabro-workflow/src/operations/fork.rs` +- Modify: `lib/crates/fabro-types/src/run_event/mod.rs` +- Modify: `lib/crates/fabro-types/src/run_event/misc.rs` +- Modify: `lib/crates/fabro-store/src/run_state.rs` +- Modify: `lib/crates/fabro-server/src/server.rs` +- Modify: `lib/crates/fabro-server/src/server/handler/steer.rs` +- Test: `lib/crates/fabro-workflow/src/event/convert.rs` +- Test: `lib/crates/fabro-store/src/run_state.rs` +- Test: `lib/crates/fabro-server/src/server/tests.rs` + +- [ ] **Step 1: Write failing event conversion tests** + +Add tests proving each event maps to stored event names: + +```text +agent.acp.started +agent.acp.completed +agent.acp.cancelled +agent.acp.timed_out +``` + +Also assert converted/stored ACP events carry `node_id`, `stage_id`, and visit-derived stage scope fields just like `agent.cli.*`. This catches missing `stored_fields.rs` wiring, which otherwise makes run projection updates fail to find the active stage. + +- [ ] **Step 2: Write failing projection tests** + +Add store tests proving: + +- `agent.acp.started` sets `stage.provider_used.mode == "acp"` +- provider, model, and command are preserved +- `agent.acp.completed` sets stage output to aggregated text/stderr payload +- cancelled and timed out terminal events set `CommandTermination::{Cancelled,TimedOut}` + +- [ ] **Step 3: Write failing fork and steerability tests** + +Add a fork replay test proving ACP provider metadata survives replay: + +- `agent.acp.started` is included by `replay_event_for_fork_projection` +- `agent.acp.cancelled` and `agent.acp.timed_out` are included for terminal metadata, matching the existing CLI terminal-event behavior + +Add server tests proving an active ACP stage is non-steerable: + +- after `agent.acp.started`, `/runs/{id}/steer` returns a conflict when no API-mode session is active +- after `agent.acp.completed`, `agent.acp.cancelled`, `agent.acp.timed_out`, `stage.completed`, or `stage.failed`, the non-steerable active-stage marker is cleared + +- [ ] **Step 4: Run tests to verify they fail** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(agent_acp)' && \ +ulimit -n 4096 && cargo nextest run -p fabro-store -E 'test(agent_acp)' && \ +ulimit -n 4096 && cargo nextest run -p fabro-server -E 'test(acp.*steer|steer.*acp)' +``` + +Expected: FAIL because event variants do not exist. + +- [ ] **Step 5: Implement event variants** + +Use fields parallel to CLI events, with ACP-specific additions, and wire `AgentAcpBackend` to emit them around `fabro_acp::run_acp_turn(...)`: + +```rust +AgentAcpStarted { + node_id: String, + visit: u32, + mode: String, // always "acp" + provider: String, + model: String, + command: String, +} + +AgentAcpCompleted { + node_id: String, + stdout: String, // aggregated ACP response text + stderr: String, + stop_reason: String, + duration_ms: u64, +} +``` + +Cancelled/timed-out events mirror CLI terminal payloads. + +- [ ] **Step 6: Implement stored fields, projection, and fork replay logic** + +Update `stored_fields.rs` so every `AgentAcp*` event gets the same node/stage fields as its `AgentCli*` counterpart. Without this, `stage_at_current_visit` and `stage_at_stored_or_visit` cannot reliably attach ACP event data to the active stage. + +Do not reuse `provider_used_from_agent_cli_started`; create `provider_used_from_agent_acp_started` so event names and mode cannot drift. + +Update `operations/fork.rs` to replay `AgentAcpStarted`, `AgentAcpCancelled`, and `AgentAcpTimedOut` for fork projections, mirroring the existing CLI started/cancelled/timed-out replay behavior. Do not add `AgentAcpCompleted` unless the implementation also intentionally changes the existing CLI completed replay policy. + +- [ ] **Step 7: Implement server steerability tracking** + +Treat ACP-backed stages as active non-steerable agent stages while they are running: + +- add `EventBody::AgentAcpStarted(_)` to the same active-stage set currently used for CLI-backed stages, or rename the set to a neutral `active_non_steerable_agent_stages` if the surrounding code becomes clearer +- remove the stage on `AgentAcpCompleted`, `AgentAcpCancelled`, `AgentAcpTimedOut`, `StageCompleted`, and `StageFailed` +- update the conflict message/code only if needed to avoid saying "CLI-mode" for an ACP-only active stage; tests should assert the API returns a clear non-steerable-agent conflict + +- [ ] **Step 8: Run tests to verify they pass** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(agent_acp)' && \ +ulimit -n 4096 && cargo nextest run -p fabro-store -E 'test(agent_acp)' && \ +ulimit -n 4096 && cargo nextest run -p fabro-server -E 'test(acp.*steer|steer.*acp)' +``` + +Expected: PASS. + +- [ ] **Step 9: Refactor and verify** + +Check event JSON field names against existing CLI event style. Do not expose raw credentials, environment variables, full JSON-RPC logs, or prompt contents in events. + +Run: + +```bash +cargo build -p fabro-types -p fabro-workflow -p fabro-store -p fabro-server +``` + +Expected: PASS. + +- [ ] **Step 10: Commit** + +```bash +git add lib/crates/fabro-workflow/src/event lib/crates/fabro-workflow/src/operations/fork.rs lib/crates/fabro-types/src/run_event lib/crates/fabro-store/src/run_state.rs lib/crates/fabro-server/src/server.rs lib/crates/fabro-server/src/server/handler/steer.rs lib/crates/fabro-server/src/server/tests.rs +git commit -m "feat: project ACP backend events" +``` + +## Task 7: Wire ACP Backend Into Pipeline Initialization + +**Files:** +- Modify: `lib/crates/fabro-workflow/src/pipeline/initialize.rs` +- Test: `lib/crates/fabro-workflow/src/pipeline/initialize.rs` +- Test: `lib/crates/fabro-workflow/tests/it/integration.rs` + +- [ ] **Step 1: Write failing initialization test** + +Add a test that builds a graph with `backend="acp"` and verifies the initialized registry can resolve the node and route to an ACP backend when credentials exist. Use a stub backend or fake ACP runner where needed; do not require live provider credentials. + +- [ ] **Step 2: Run test to verify it fails** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(initialize.*acp) | test(backend_router_delegates_to_acp_for_acp_node)' +``` + +Expected: FAIL because initialization still constructs only API and CLI. + +- [ ] **Step 3: Implement initialization wiring** + +In `build_registry`, construct: + +```rust +let acp = AgentAcpBackend::new(model.clone(), provider, cli_resolver.clone()) + .with_tool_env_provider(tool_env_provider, github_token_refresh_managed); +Some(Box::new(BackendRouter::new(Box::new(api), cli, acp))) +``` + +If the resolver type is not cloneable, restructure so CLI and ACP each receive their own resolver handle from the same source. Do not make ACP fall back to host env when a vault resolver is available. + +- [ ] **Step 4: Run tests to verify they pass** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(initialize.*acp) | test(backend_router_delegates_to_acp_for_acp_node)' +``` + +Expected: PASS. + +- [ ] **Step 5: Refactor and verify** + +Ensure dry-run behavior is unchanged: dry-run builds no real backend and simulates all LLM handlers. + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(dry_run) | test(router_)' +``` + +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add lib/crates/fabro-workflow/src/pipeline/initialize.rs lib/crates/fabro-workflow/tests/it/integration.rs +git commit -m "feat: initialize ACP workflow backend" +``` + +## Task 8: Add Black-Box Workflow Coverage With A Fake ACP Agent + +**Files:** +- Create: `lib/crates/fabro-cli/tests/fixtures/acp/fake_acp_agent.rs` or equivalent checked-in script fixture +- Modify: `lib/crates/fabro-cli/tests/it/workflow/real_cli.rs` or create `lib/crates/fabro-cli/tests/it/workflow/acp.rs` +- Test: `lib/crates/fabro-cli/tests/it/workflow/` + +- [ ] **Step 1: Write failing black-box test** + +Create a temp workflow with an ACP-backed agent node: + +```dot +digraph { + work [type="agent", backend="acp", provider="openai", model="fake-acp", prompt="write hello.txt"] +} +``` + +Use a fake ACP command fixture that: + +- responds to `initialize` +- responds to `session/new` +- on `session/prompt`, writes `hello.txt` in cwd +- emits two `agent_message_chunk` updates +- returns `stopReason: "end_turn"` + +The test must assert: + +- run succeeds +- response text contains concatenated chunks +- `hello.txt` is included in `files_touched` +- provider projection has `mode == "acp"` + +- [ ] **Step 2: Run test to verify it fails** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-cli -E 'test(acp_backend_workflow)' +``` + +Expected: FAIL before CLI/workflow wiring is complete. + +- [ ] **Step 3: Use the `acp_command` override path** + +Set the test node's `acp_command` to the checked-in fake agent command. The override was added in Task 2 and wired through the adapter in Task 5; this black-box test proves it works through real workflow parsing and execution. Do not add global config schema for this cutover. + +- [ ] **Step 4: Run test to verify it passes** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-cli -E 'test(acp_backend_workflow)' +``` + +Expected: PASS. + +- [ ] **Step 5: Refactor and verify** + +Keep the fake agent deterministic and dependency-light. Prefer a Rust test helper binary if it avoids shell quoting differences across macOS/Linux. + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-cli -E 'test(acp_backend_workflow) | test(full_pipeline_with_cli_backend_node)' +``` + +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add lib/crates/fabro-cli/tests +git commit -m "test: cover ACP backend workflow execution" +``` + +## Task 9: Update Documentation + +**Files:** +- Modify: `docs/public/reference/dot-language.mdx` +- Modify: `docs/public/core-concepts/agents.mdx` +- Modify: any docs found by `rg -n "backend=.*cli|backend: cli|backend.*api|CLI backend|cli mode" docs/public lib/crates -g '*.md' -g '*.mdx'` + +- [ ] **Step 1: Find stale backend docs** + +Run: + +```bash +rg -n "backend=.*cli|backend: cli|backend.*api|CLI backend|cli mode|one_shot" docs/public lib/crates -g '*.md' -g '*.mdx' +``` + +Expected: list current docs that mention only API/CLI or imply prompt nodes use CLI. + +- [ ] **Step 2: Update docs** + +Document: + +- `backend` values are `api`, `cli`, and `acp` +- default is `api` unless model-specific routing says otherwise +- `cli` is legacy and may be replaced by ACP later +- `acp` runs an ACP agent via stdio inside supported sandboxes +- stable ACP does not portably accept model selection; Fabro records model metadata but command choice controls model behavior for now +- `acp_command` advanced override if implemented in Task 8 +- Daytona ACP limitation if still unsupported + +- [ ] **Step 3: Verify docs references** + +Run: + +```bash +rg -n "backend=.*cli|backend: cli|backend.*api|CLI backend|cli mode|ACP" docs/public lib/crates -g '*.md' -g '*.mdx' +``` + +Expected: no stale claim that prompt nodes support CLI, no missing ACP mention in backend reference. + +- [ ] **Step 4: Commit** + +```bash +git add docs/public +git commit -m "docs: document ACP workflow backend" +``` + +## Task 10: Final Verification + +**Files:** +- Verify entire changed set + +- [ ] **Step 1: Run targeted ACP tests** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-acp +ulimit -n 4096 && cargo nextest run -p fabro-sandbox -E 'test(stdio)' +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_) | test(acp_backend) | test(agent_acp) | test(initialize.*acp)' +ulimit -n 4096 && cargo nextest run -p fabro-validate -E 'test(backend_valid)' +ulimit -n 4096 && cargo nextest run -p fabro-store -E 'test(agent_acp)' +ulimit -n 4096 && cargo nextest run -p fabro-server -E 'test(acp.*steer|steer.*acp)' +ulimit -n 4096 && cargo nextest run -p fabro-cli -E 'test(acp_backend_workflow)' +``` + +Expected: all PASS. + +- [ ] **Step 2: Run regression tests named in the accepted strategy** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run -p fabro-workflow -E 'test(router_uses_cli_for_backend_attr) | test(router_uses_api_by_default) | test(backend_router_delegates_to_cli_for_cli_node) | test(backend_router_delegates_to_api_for_normal_node) | test(backend_router_delegates_to_cli_for_backend_attr) | test(full_pipeline_with_cli_backend_node) | test(stylesheet_backend_property_routes_to_cli) | test(cli_backend_run_writes_prompt_and_calls_exec) | test(cli_backend_run_with_codex_provider) | test(parse_real_codex_ndjson)' +ulimit -n 4096 && cargo nextest run -p fabro-mcp -E 'test(stdio_client_initialize_and_list_tools) | test(stdio_client_call_tool_echo) | test(connection_manager_stdio_roundtrip)' +``` + +Expected: all PASS. + +- [ ] **Step 3: Run broader checks** + +Run: + +```bash +cargo build --workspace +cargo +nightly-2026-04-14 fmt --check --all +cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings +``` + +Expected: all PASS. + +- [ ] **Step 4: Inspect final diff** + +Run: + +```bash +git status --short +git diff --stat main...HEAD +git diff --name-only main...HEAD +``` + +Expected: only ACP backend, sandbox stdio, event/projection, tests, and docs changes. + +- [ ] **Step 5: Commit final fixes if needed** + +If verification required code or docs fixes: + +```bash +git add +git commit -m "fix: complete ACP backend verification" +``` + +Expected: clean worktree. + +## Regression Risks + +- **Sandbox isolation bypass:** using `agent-client-protocol-tokio::AcpAgent` directly would spawn host processes. The implementation must instead run ACP stdio through the sandbox abstraction. +- **Prompt-node host execution:** ACP prompt nodes cannot be implemented against the old `CodergenBackend::one_shot` signature because it lacks sandbox/cancel-token access. The trait and `PromptHandler` call site must change, and API-backed one-shot implementations must ignore the new parameters. +- **SDK transport mismatch:** the ACP SDK expects futures-style line/byte streams; Fabro's sandbox API must expose stdin/stdout in a form `fabro-acp` can adapt to `agent_client_protocol::Lines` or `ByteStreams`. A write-line/read-line-only sandbox API is insufficient if it cannot be converted into a real transport. +- **Command parsing drift:** do not validate `acp_command` with `AcpAgent::from_str` and then execute the raw string. JSON stdio configs would be accepted but fail at runtime, and shell-word parsing would not protect sandbox command construction. +- **Missing Node runtime:** default ACP commands use `npx`. Without the shared Node bootstrap, fresh local/Docker sandboxes that currently work with `backend="cli"` may fail immediately with `npx: command not found`. +- **Crate cycle:** `fabro-acp` cannot implement workflow's `CodergenBackend` directly without making `fabro-workflow` and `fabro-acp` depend on each other. Keep protocol implementation in `fabro-acp` and the trait adapter in workflow. +- **PTY corruption:** ACP JSON-RPC must not use terminal sessions. Docker stdio uses `tty=false`; Daytona remains unsupported until raw stdio exists. +- **Permission deadlock:** ACP agents can send `session/request_permission` before performing edits. Without an automatic permission handler, real adapters may fail with method-not-found or stall even though legacy CLI mode already grants full-auto permissions. +- **Docker orphaned processes:** Bollard does not provide a simple kill-exec primitive. Docker stdio must use a controlled wrapper/stop-file or equivalent process-group cleanup so timeout and cancellation do not leave ACP adapters running inside the container. +- **Decorator capability loss:** `delegate_sandbox!`, `WorktreeSandbox`, read/write guards, and test-support sandboxes must forward `spawn_stdio_process`; otherwise ACP works on the base local sandbox but fails once wrapped by normal workflow setup. +- **Prompt backend ambiguity:** `backend="acp"` on prompt nodes must route to ACP or fail clearly. It must not silently use API. +- **Model drift:** stable ACP does not standardize model selection. Do not pretend node `model` was sent to the ACP agent unless the implementation actually supports it through an explicit command/config mechanism. +- **Event projection drift:** ACP events must set `mode="acp"` independently from CLI projection helpers. +- **Missing stored fields or fork replay:** ACP events must populate stage-scoped stored fields and fork replay rules. Otherwise provider metadata may be emitted correctly but fail to attach to the stage projection or disappear after a fork. +- **Incorrect steerability while ACP is running:** ACP-backed stages are not API-mode steerable in this cutover. The server must treat active ACP stages as non-steerable agent stages so steer requests do not get buffered as if an API session might appear. +- **Credential regression:** ACP must use the same credential resolver path as CLI mode so vault-backed installs do not fall back to missing host env vars. diff --git a/docs/plans/2026-05-11-add-fabro-mcp-server-test-plan.md b/docs/plans/2026-05-11-add-fabro-mcp-server-test-plan.md new file mode 100644 index 000000000..d90a41459 --- /dev/null +++ b/docs/plans/2026-05-11-add-fabro-mcp-server-test-plan.md @@ -0,0 +1,292 @@ +# Fabro MCP Server Test Plan + +## Harness Requirements + +The agreed testing strategy still holds after reading the implementation plan. The plan narrows the tool contract to five Devin-shaped run tools and requires the implementation to live in a new `fabro-mcp-server` crate, but it does not add paid APIs, live LLM calls, external infrastructure, or browser/UI behavior. The highest-value evidence remains a real `fabro mcp start` subprocess driven over stdio and backed by Fabro's real local test server/auth harness. + +1. **Deterministic MCP stdio fixture** + - **Does:** constructs the exact command, environment, and cwd used to spawn `env!("CARGO_BIN_EXE_fabro") mcp start`. + - **Exposes:** `command: Vec`, `env: HashMap`, and `current_dir: PathBuf` usable by both `fabro_mcp::client::McpClient` and raw `std::process::Command` tests. + - **Complexity:** low. Add a narrow helper in `lib/crates/fabro-cli/tests/it/cmd/mcp.rs`; if needed, add `fabro_test::isolated_env(home_dir)` to mirror `apply_test_isolation`. + - **Tests depending on it:** 5, 6, 7, 8, 9, 10, 15, 16, 17. + +2. **MCP tool-call assertion helpers** + - **Does:** calls a named MCP tool, asserts tool success or tool error, extracts `structured_content`, and verifies fallback text is concise rather than a JSON dump. + - **Exposes:** `call_tool_json(...)`, `call_tool_error_text(...)`, and normalization helpers for run IDs, timestamps, paths, event IDs, cursors, durations, and elapsed times. + - **Complexity:** low to medium. Keep it local to `cmd/mcp.rs` unless more than one test file needs it. + - **Tests depending on it:** 8, 9, 10, 11, 12, 13, 15, 16, 17. + +3. **Real authenticated Fabro server fixture** + - **Does:** starts `RealAuthHarness::start_with_dev_token(...)`, seeds CLI dev-token auth into the test home, creates dry-run workflows through public CLI/MCP/API surfaces, and shuts down the server. + - **Exposes:** API target URL, persisted auth entry, HTTP client/server-visible state checks, and workflow fixture paths. + - **Complexity:** medium, mostly reuse existing `lib/crates/fabro-cli/tests/it/support/auth_harness.rs`. + - **Tests depending on it:** 8, 10, 11, 12, 13, 14, 17. + +## Test Plan + +1. **`fabro mcp` help exposes the MCP namespace** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness through existing `fabro_snapshot!` + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp --help`. + - **Expected outcome:** stdout snapshots a `Model Context Protocol server` namespace with `start`, `config`, and `init` subcommands; stderr is empty; exit status is 0. Source of truth: user request for `fabro mcp start`, `fabro mcp config`, `fabro mcp init `, and implementation plan CLI contract. + - **Interactions:** clap command tree, global CLI flags, snapshot filters. + +2. **`fabro mcp start --help` documents stdio startup options** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness through `fabro_snapshot!` + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp start --help`. + - **Expected outcome:** stdout snapshots usage `fabro mcp start [OPTIONS]` with `--server ` and `--storage-dir `; stderr is empty; exit status is 0. Source of truth: implementation plan CLI contract. + - **Interactions:** clap flattening for `ServerConnectionArgs`. + +3. **`fabro mcp config --help` documents config rendering options** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness through `fabro_snapshot!` + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp config --help`. + - **Expected outcome:** stdout snapshots usage and the same connection override flags as `start`; stderr is empty; exit status is 0. Source of truth: implementation plan CLI contract. + - **Interactions:** clap command help and global CLI flags. + +4. **`fabro mcp init --help` documents supported agent selection** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness through `fabro_snapshot!` + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp init --help`. + - **Expected outcome:** stdout snapshots required `` with supported values `claude`, `cursor`, and `windsurf`; exit status is 0. Source of truth: user request and implementation plan supported-agent contract. + - **Interactions:** clap value enum rendering. + +5. **`fabro mcp config` prints generic MCP client JSON** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness plus structured JSON parsing + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp config`; parse stdout as JSON. + - **Expected outcome:** stdout is valid JSON with `mcpServers.fabro.command == "fabro"` and `args == ["mcp", "start"]`; stderr is empty; exit status is 0. Source of truth: Daytona-shaped user request and implementation plan config JSON contract. + - **Interactions:** config rendering, stdout contract for a non-stdio command. + +6. **`fabro mcp config` preserves connection flags in generated startup args** + - **Type:** integration + - **Disposition:** new + - **Harness:** output capture harness plus structured JSON parsing + - **Preconditions:** isolated `TestContext`; no auth or server required. + - **Actions:** run `fabro mcp config --server https://example.test/api/v1 --storage-dir /tmp/fabro-mcp-storage`; parse stdout as JSON. + - **Expected outcome:** JSON contains `args == ["mcp", "start", "--server", "https://example.test/api/v1", "--storage-dir", "/tmp/fabro-mcp-storage"]`; stderr is empty; exit status is 0. Source of truth: implementation plan examples for flag preservation. + - **Interactions:** CLI argument forwarding into MCP client config. + +7. **`fabro mcp init ` writes idempotent config without clobbering unrelated keys** + - **Type:** integration + - **Disposition:** new + - **Harness:** direct filesystem artifact assertion in isolated home + - **Preconditions:** isolated `TestContext`; pre-existing Cursor config with `mcpServers.other` and unrelated top-level key. + - **Actions:** run `fabro mcp init cursor --server https://example.test/api/v1` twice; read `~/.cursor/mcp.json`. + - **Expected outcome:** parsed JSON preserves unrelated keys and existing `mcpServers.other`, contains exactly one `mcpServers.fabro` entry with command `fabro` and expected args, and the second run does not duplicate or reorder into an invalid shape. Source of truth: implementation plan idempotent config merge contract. + - **Interactions:** filesystem directory creation, JSON merge/write, test home isolation. + +8. **`fabro mcp init` writes each supported agent path** + - **Type:** integration + - **Disposition:** new + - **Harness:** direct filesystem artifact assertion in isolated home + - **Preconditions:** isolated `TestContext`; no existing Claude, Cursor, or Windsurf config. + - **Actions:** run `fabro mcp init claude`, `fabro mcp init cursor`, and `fabro mcp init windsurf` in separate contexts; read the platform-specific config file for each. + - **Expected outcome:** each config file exists at the path named by the implementation plan and contains `mcpServers.fabro` with `command: "fabro"` and `args: ["mcp", "start"]`. Source of truth: implementation plan agent path contract. + - **Interactions:** platform-specific path selection, filesystem writes. + +9. **`fabro mcp init` rejects invalid existing config without overwrite** + - **Type:** boundary + - **Disposition:** new + - **Harness:** output capture and filesystem artifact assertion + - **Preconditions:** isolated `TestContext`; Cursor config file contains invalid JSON bytes. + - **Actions:** run `fabro mcp init cursor`; read the same file after failure. + - **Expected outcome:** command exits non-zero with a clear error that includes the config path; the file content is byte-for-byte unchanged. Source of truth: implementation plan invalid JSON failure contract and error-handling strategy. + - **Interactions:** JSON parsing, write avoidance on error, CLI error rendering. + +10. **`fabro mcp start` initializes over stdio and lists the five run tools** + - **Type:** scenario + - **Disposition:** new + - **Harness:** interaction harness using deterministic MCP stdio fixture and `fabro_mcp::client::McpClient` + - **Preconditions:** isolated `TestContext`; no auth; no live Fabro server. + - **Actions:** spawn `fabro mcp start`; perform MCP `initialize`; call `tools/list`. + - **Expected outcome:** initialize succeeds without auth/server connectivity; `tools/list` returns exactly `fabro_run_create`, `fabro_run_search`, `fabro_run_interact`, `fabro_run_gather`, and `fabro_run_events`, each with an input schema. Source of truth: MCP lifecycle/tools spec as captured in the agreed strategy and implementation plan exact tool list. + - **Interactions:** `rmcp` stdio transport, existing `fabro-mcp` client crate, child process lifecycle. + +11. **`fabro mcp start` reserves stdout for JSON-RPC only** + - **Type:** regression + - **Disposition:** new + - **Harness:** raw subprocess stdio harness + - **Preconditions:** isolated `TestContext`; no auth; no live Fabro server. + - **Actions:** spawn `fabro mcp start`; write a JSON-RPC `initialize` request to stdin; read the first stdout line. + - **Expected outcome:** first stdout line parses as JSON and has `jsonrpc: "2.0"`; no leading human log/help text appears on stdout; stderr may contain logs. Source of truth: MCP stdio transport contract and implementation plan stdout invariant. + - **Interactions:** CLI logging initialization, raw process pipes, JSON-RPC framing. + +12. **MCP startup and tool discovery are fast without auth or server** + - **Type:** invariant + - **Disposition:** new + - **Harness:** interaction harness plus timing assertion + - **Preconditions:** isolated `TestContext`; no auth; no live Fabro server. + - **Actions:** measure elapsed time for spawning `fabro mcp start`, initializing, and calling `tools/list`. + - **Expected outcome:** operation completes under a generous smoke threshold, initially 2 seconds unless CI evidence requires a documented adjustment; all five tools are listed. Source of truth: agreed testing strategy performance smoke and implementation plan lazy API connection invariant. + - **Interactions:** process startup, `rmcp` initialization, tool schema generation. + +13. **`fabro_run_create` creates and starts a real dry-run using persisted CLI auth** + - **Type:** scenario + - **Disposition:** new + - **Harness:** interaction harness plus real authenticated Fabro server fixture + - **Preconditions:** `RealAuthHarness::start_with_dev_token(...)`; dev-token auth seeded into isolated home for the harness target; checked-in `simple.fabro` fixture installed. + - **Actions:** spawn `fabro mcp start --server `; call `fabro_run_create` with one run using `workflow`, `dry_run: true`, `auto_approve: true`, and label `source=mcp-test`. + - **Expected outcome:** tool result is not an MCP error; `structured_content.runs[0]` includes a run id, workflow, `started: true`, and status; fallback text exists and does not start with `{` or `[`; server-visible state contains the created run. Source of truth: user request for run-management MCP tools, implementation plan create semantics, OpenAPI `POST /api/v1/runs`, and `POST /api/v1/runs/{id}/start`. + - **Interactions:** persisted CLI auth store, Fabro API client, manifest builder/validation, run engine dry-run path. + +14. **`fabro_run_search` filters, paginates, and includes archived runs by default** + - **Type:** integration + - **Disposition:** new + - **Harness:** interaction harness plus real authenticated Fabro server fixture + - **Preconditions:** authenticated MCP server; at least two MCP-created dry-run runs with distinct labels; one terminal run archived through API or MCP. + - **Actions:** call `fabro_run_search` with `run_ids`, `workflow`, `labels`, `status`, `archived`, `first`, and `after` combinations. + - **Expected outcome:** results are normalized run summaries; filters include only matching runs; `first` limits page size and returns an opaque cursor when more results exist; archived runs appear unless `archived: false` is supplied. Source of truth: implementation plan search semantics and OpenAPI list-runs include-archived behavior adapted by the plan. + - **Interactions:** server run listing, status string normalization, timestamp/date parsing, cursor handling. + +15. **`fabro_run_interact get/start/message/cancel` uses selector resolution and server APIs** + - **Type:** integration + - **Disposition:** new + - **Harness:** interaction harness plus mocked HTTP server for precise API call assertions + - **Preconditions:** isolated `TestContext`; HTTP mock server with `/api/v1/runs/resolve`, `/runs/{id}`, `/state`, `/start`, `/steer`, and `/cancel` endpoints; CLI auth seeded if the mock requires auth. + - **Actions:** call `fabro_run_interact` with actions `get`, `start`, `message` with `interrupt: true`, and `cancel`, using a workflow-name selector rather than the exact run id. + - **Expected outcome:** each action first resolves the selector through `/runs/resolve`; calls the matching endpoint; returns a structured object with `run_id`, `action`, and action-specific `result`; tool errors are not produced for mocked successful API responses. Source of truth: implementation plan interact semantics and OpenAPI operation descriptions for retrieve, state, start, steer, and cancel. + - **Interactions:** run selector semantics, API error conversion, structured content projection. + +16. **`fabro_run_interact archive/unarchive` changes real server-visible archived state** + - **Type:** scenario + - **Disposition:** new + - **Harness:** interaction harness plus real authenticated Fabro server fixture + - **Preconditions:** authenticated MCP server; completed dry-run created through MCP or public CLI. + - **Actions:** call `fabro_run_interact` with `archive`; call `fabro_run_search` with `archived: true`; call `fabro_run_interact` with `unarchive`; call `fabro_run_search` with `archived: false`. + - **Expected outcome:** archive action succeeds for the terminal run; archived search shows the run; unarchive action succeeds; unarchived search shows the run as terminal and not archived. Source of truth: implementation plan interact actions and OpenAPI archive/unarchive contracts. + - **Interactions:** archive state transitions, list/search visibility, server-side idempotence. + +17. **`fabro_run_interact get_questions/answer` maps answer JSON to the API contract** + - **Type:** integration + - **Disposition:** new + - **Harness:** interaction harness plus mocked HTTP server for endpoint/body assertions + - **Preconditions:** isolated `TestContext`; HTTP mock server returns pending questions and accepts answer submissions. + - **Actions:** call `fabro_run_interact` with `get_questions`; call `answer` using representative payloads: `true`, `false`, string text, `{ "option": "a" }`, `{ "options": ["a", "b"] }`, and `{ "text": "hello" }`. + - **Expected outcome:** `get_questions` returns the API question list projection; `answer` sends `SubmitAnswerRequest` wire shapes with `kind: yes`, `no`, `text`, `selected`, and `multi_selected`, and returns a successful structured action result. Source of truth: implementation plan answer mapping and `lib/crates/fabro-api/tests/submit_answer_request_round_trip.rs`. + - **Interactions:** generated API type shape, JSON body serialization, human-in-the-loop endpoints. + +18. **`fabro_run_gather` waits for terminal runs and returns current state on timeout** + - **Type:** scenario + - **Disposition:** new + - **Harness:** interaction harness plus real authenticated Fabro server fixture + - **Preconditions:** authenticated MCP server; one completed dry-run and one submitted/non-terminal run available. + - **Actions:** call `fabro_run_gather` on the completed run; call it on the non-terminal run with `timeout_seconds: 1` and `poll_interval_seconds: 5`. + - **Expected outcome:** completed run result has `timed_out: false` and terminal status; timeout case returns a successful structured result with `timed_out: true`, current run summary, and bounded elapsed wall time rather than an MCP/process error. Source of truth: implementation plan gather semantics and agreed performance/timeout strategy. + - **Interactions:** selector resolution, polling loop, server retrieve endpoint, terminal status classification. + +19. **`fabro_run_events` lists, details, searches, filters, paginates, and truncates events** + - **Type:** integration + - **Disposition:** new + - **Harness:** interaction harness plus real authenticated Fabro server fixture + - **Preconditions:** authenticated MCP server; completed dry-run with stored events. + - **Actions:** call `fabro_run_events` with `action: "list"` and `first`; call `details` with returned event ids; call `search` with a known event-name substring; call filters for `event_types`, `categories`, `direction: "desc"`, `after`, `offset`, `limit`, and a small `max_content_length`. + - **Expected outcome:** returned events belong to the run; list ordering and pagination match requested parameters; details returns only requested event ids; search returns serialized events containing the query; category filtering uses event-name prefix; oversized serialized payloads are truncated with `truncated: true`; `next_cursor` is derived from the last returned sequence. Source of truth: implementation plan events semantics and OpenAPI `GET /api/v1/runs/{id}/events`. + - **Interactions:** event store pagination, event-name/category derivation, JSON serialization/truncation. + +20. **Local validation errors happen before auth or network lookup and do not stop the server** + - **Type:** boundary + - **Disposition:** new + - **Harness:** interaction harness with `--server http://127.0.0.1:9` and no auth + - **Preconditions:** isolated `TestContext`; no auth entry; unreachable server URL. + - **Actions:** call `fabro_run_gather` with 51 run ids; call `tools/list`; call `fabro_run_interact` action `message` without `message`; call `tools/list` again. + - **Expected outcome:** each invalid tool call returns an MCP tool error mentioning the invalid field (`run_ids` or `message`); no auth guidance or connection error masks the local validation failure; subsequent `tools/list` succeeds. Source of truth: implementation plan validate-before-client invariant and MCP tool-error contract. + - **Interactions:** parameter validation, lazy client initialization, MCP service liveness after errors. + +21. **Auth failures use existing Fabro login guidance and remain tool errors** + - **Type:** boundary + - **Disposition:** new + - **Harness:** interaction harness with protected real or mocked API target + - **Preconditions:** isolated `TestContext`; no saved auth for the target; server requires auth. + - **Actions:** spawn `fabro mcp start --server `; call a valid read tool such as `fabro_run_search`. + - **Expected outcome:** call returns an MCP tool error, not process exit; error text includes `Run \`fabro auth login\` to authenticate.`; subsequent `tools/list` still succeeds. Source of truth: user request for no separate MCP auth and implementation plan auth invariant. + - **Interactions:** auth store lookup, client connection, error classification/rendering. + +22. **Invalid create inputs are rejected with field-specific tool errors** + - **Type:** boundary + - **Disposition:** new + - **Harness:** interaction harness with no auth and unreachable server + - **Preconditions:** isolated `TestContext`; no auth entry. + - **Actions:** call `fabro_run_create` with empty `runs`, with 51 runs, and with `inputs` containing a null value. + - **Expected outcome:** each call returns an MCP tool error naming the invalid field/key before any auth/server error; server remains alive for a subsequent `tools/list`. Source of truth: implementation plan create validation and JSON-to-TOML null rejection. + - **Interactions:** schema/validation layer, JSON-to-TOML conversion. + +23. **Run tool successes always include structured content and concise text** + - **Type:** invariant + - **Disposition:** new + - **Harness:** MCP tool-call assertion helpers reused by scenario tests + - **Preconditions:** any successful calls from tests 13, 14, 16, 18, and 19. + - **Actions:** for each successful call, inspect `CallToolResult`. + - **Expected outcome:** `structured_content` is present; at least one text content item is present; text content is short and does not begin with `{` or `[`; `is_error` is absent or false. Source of truth: implementation plan successful tool-result invariant. + - **Interactions:** `rmcp::model::CallToolResult` construction and MCP client display fallback. + +24. **Pure conversion helpers cover JSON-to-TOML and answer-request mapping** + - **Type:** unit + - **Disposition:** new + - **Harness:** `cargo nextest run -p fabro-mcp-server run_tools` + - **Preconditions:** none beyond crate compilation. + - **Actions:** call conversion helpers directly for strings, bools, integers, floats, arrays, objects, null input, and every supported answer payload shape. + - **Expected outcome:** JSON-compatible input values map to equivalent `toml::Value`; null returns an error naming the key; answer payloads serialize to `SubmitAnswerRequest` wire JSON with documented `kind` values; unsupported answer objects return a tool error. Source of truth: implementation plan conversion requirements and `fabro-api` submit-answer round-trip tests. + - **Interactions:** serde, generated API types, conversion error text. + +25. **Existing MCP client crate behavior is not regressed** + - **Type:** regression + - **Disposition:** existing + - **Harness:** existing `fabro-mcp` crate tests + - **Preconditions:** repository builds with the new `fabro-mcp-server` crate added. + - **Actions:** run `cargo nextest run -p fabro-mcp`. + - **Expected outcome:** existing stdio client initialize/list/call tests pass. Source of truth: existing automated evidence and implementation plan decision to keep `fabro-mcp` as the external MCP client crate. + - **Interactions:** workspace dependency feature unification for `rmcp`, existing client transport behavior. + +26. **Relevant existing CLI run/auth regressions still pass** + - **Type:** regression + - **Disposition:** existing + - **Harness:** existing `fabro-cli` integration tests + - **Preconditions:** implementation complete. + - **Actions:** run the existing tests matching `scenario::auth::auth_login_refresh_logout_flow`, `scenario::lifecycle::dry_run_create_start_attach_works_with_default_run_lookup`, and `cmd::ps::ps_explicit_local_tcp_target_uses_auth_store`; if names drift, list tests and run the corresponding auth/lifecycle/local-target checks. + - **Expected outcome:** all selected tests pass unchanged. Source of truth: agreed strategy existing automated evidence and user requirement that MCP reuse CLI auth/config behavior. + - **Interactions:** auth refresh/logout, local server run lifecycle, server target resolution. + +27. **Final MCP command contract and workspace checks pass** + - **Type:** regression + - **Disposition:** extend + - **Harness:** repository command checks + - **Preconditions:** all feature implementation and snapshots complete. + - **Actions:** run `cargo nextest run -p fabro-cli --test it cmd::mcp`, `cargo +nightly-2026-04-14 fmt --check --all`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`, `ulimit -n 4096 && cargo nextest run --workspace`, and `cargo insta pending-snapshots`. + - **Expected outcome:** MCP command tests pass; formatting and clippy pass; workspace tests pass; no pending snapshots remain unless explicitly inspected and accepted for this feature. Source of truth: repository `AGENTS.md` build/test commands and snapshot policy. + - **Interactions:** entire workspace, rustfmt/clippy pinned nightly, nextest parallelism and file descriptor limit. + +## Coverage Summary + +Covered action space: + +- CLI executable commands: `fabro mcp --help`, `fabro mcp start --help`, `fabro mcp config --help`, `fabro mcp init --help`, `fabro mcp config`, `fabro mcp config --server --storage-dir`, and `fabro mcp init claude|cursor|windsurf`. +- MCP protocol actions: stdio process startup, `initialize`, `tools/list`, and `tools/call`. +- MCP tool actions: `fabro_run_create`; `fabro_run_search`; `fabro_run_interact` actions `get`, `start`, `message`, `cancel`, `archive`, `unarchive`, `get_questions`, `answer`; `fabro_run_gather`; `fabro_run_events` actions `list`, `details`, and `search`. +- Error and boundary behavior: invalid local parameters, too many run ids, null input conversion, missing action fields, unsupported answer shapes, invalid agent config JSON, missing auth, unreachable server after local validation, timeout expiry, and service liveness after tool errors. +- Integration boundaries: CLI auth store reuse, Fabro API client, real local Fabro server, run manifest construction/validation, event store, generated API answer types, and existing `fabro-mcp` client crate. +- Performance smoke: initialize plus `tools/list` without auth/server. + +Explicitly excluded per the agreed strategy: + +- Live LLM/provider tests. Dry-run workflows and local/mocked servers cover run-management behavior without external credentials or spend. +- Manual QA of agent apps. `init` tests assert Fabro's written config path and JSON merge contract, not whether Claude/Cursor/Windsurf accept the file in a live app. +- Browser/UI tests. This feature adds CLI and MCP stdio surfaces only. +- Differential tests against Daytona or Devin. Their docs inspired shape, but no runnable reference implementation is available or required. + +Residual risks: + +- Agent config formats may evolve externally; tests protect Fabro's chosen file/path contract only. +- MCP SDK behavior can change with `rmcp` upgrades; protocol tests and existing `fabro-mcp` tests should catch startup/list/call regressions. +- Full workspace tests may be slower and subject to local FD limits; use the documented `ulimit -n 4096` command before `cargo nextest run --workspace`. diff --git a/docs/plans/2026-05-11-add-fabro-mcp-server.md b/docs/plans/2026-05-11-add-fabro-mcp-server.md new file mode 100644 index 000000000..1e543a8a9 --- /dev/null +++ b/docs/plans/2026-05-11-add-fabro-mcp-server.md @@ -0,0 +1,1718 @@ +# Fabro MCP Server Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use trycycle-executing to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add a stdio MCP server to the `fabro` CLI, exposed as `fabro mcp start`, with `fabro mcp config` and `fabro mcp init ` support and first-class tools for managing Fabro runs. + +**Architecture:** Implement the Fabro MCP server in a new `fabro-mcp-server` crate, with `fabro-cli` owning only clap parsing and dispatch for `fabro mcp ...`. Use `rmcp` server macros and stdio transport for protocol correctness, connect lazily to the Fabro API through settings built from the same CLI auth/config inputs as existing Fabro commands, and return structured MCP tool results plus text fallbacks. Keep the existing `fabro-mcp` crate as the external MCP client/shared protocol support used by Fabro agents and tests, not as the server crate. + +**Tech Stack:** Rust, clap, tokio, new `fabro-mcp-server` crate, rmcp 1.3 stdio server transport, serde/schemars JSON schemas, fabro-client, fabro-api generated types, existing Fabro CLI integration test harness with insta snapshots. + +--- + +## File Structure + +- Create `lib/crates/fabro-mcp-server/Cargo.toml` + - New crate for the stdio MCP server implementation. Add direct `rmcp` dependency with server, macros, schemars, and stdio transport features, plus the Fabro crates needed for API access, run manifest construction, settings/auth-store resolution, and tests. The workspace already includes `lib/crates/*`, so no root workspace member edit is required. + +- Create `lib/crates/fabro-mcp-server/src/lib.rs` + - Export the server entry points and settings types consumed by `fabro-cli`: `McpServerSettings`, `McpConfigSettings`, `McpAgent`, `start(settings)`, `config_json(settings)`, and `init_agent(settings)`. + +- Create `lib/crates/fabro-mcp-server/src/server.rs` + - Own the stdio MCP service, tool registration, API client acquisition from explicit settings, and tool error shaping. + +- Create `lib/crates/fabro-mcp-server/src/run_tools.rs` + - Own run-management behavior behind the MCP tools: create/start, search, interact, gather, and events. + - This split keeps protocol boilerplate out of run semantics. + +- Create `lib/crates/fabro-mcp-server/src/config.rs` + - Own generic MCP config rendering and agent-specific config path/merge/write logic. + +- Modify `lib/crates/fabro-cli/Cargo.toml` + - Add a path dependency on the new `fabro-mcp-server` crate. `fabro-cli` should not depend directly on `rmcp` for the server implementation. + +- Modify `lib/crates/fabro-cli/src/args.rs` + - Add `McpNamespace`, `McpCommand`, `McpStartArgs`, `McpConfigArgs`, `McpInitArgs`, and `McpAgent`. + - Add `Commands::Mcp(McpNamespace)` and `Commands::name()` branch returning `mcp start`, `mcp config`, or `mcp init`. + +- Modify `lib/crates/fabro-cli/src/main.rs` + - Add `mod commands::mcp` dispatch. + - Keep `fabro mcp start` on the normal CLI logging path, which writes logs to stderr, and never write human output to stdout during stdio serving. + +- Modify `lib/crates/fabro-cli/src/commands/mod.rs` + - Export the new `mcp` command module. + +- Create `lib/crates/fabro-cli/src/commands/mcp/mod.rs` + - Own CLI dispatch for `start`, `config`, and `init`. + +- Modify `lib/crates/fabro-cli/src/commands/run/overrides.rs` + - If needed, move shared manifest override construction into a non-CLI crate or expose a small reusable helper without creating a dependency from `fabro-mcp-server` back to `fabro-cli`: + - label parsing + - goal layer construction + - execution/model/sandbox override construction + - Do not duplicate manifest override semantics in the MCP server crate. + +- Modify `lib/crates/fabro-cli/tests/it/cmd/mod.rs` + - Add `mod mcp;`. + +- Create `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` + - Add CLI help/config/init snapshots and stdio MCP integration tests. + +- Optionally modify `lib/crates/fabro-cli/tests/it/support/mod.rs` + - Add only narrow helpers for spawning `fabro mcp start` or extracting MCP text/structured output if duplication appears in `cmd/mcp.rs`. + +Before editing Rust code, read: + +- `docs/internal/testing-strategy.md` because this plan adds CLI integration tests and unit tests. +- `docs/internal/error-handling-strategy.md` because MCP tool failures convert CLI/API/auth errors into user-visible tool errors. + +## User-Visible Contract + +The CLI contract is: + +```text +fabro mcp start [--server ] [--storage-dir ] +fabro mcp config [--server ] [--storage-dir ] +fabro mcp init [--server ] [--storage-dir ] +``` + +Supported agents for the first implementation: + +```text +claude +cursor +windsurf +``` + +`fabro mcp config` emits generic MCP client JSON to stdout: + +```json +{ + "mcpServers": { + "fabro": { + "command": "fabro", + "args": ["mcp", "start"] + } + } +} +``` + +When `--server` or `--storage-dir` is passed to `config` or `init`, preserve those choices in the emitted or written `args`, for example: + +```json +{ + "mcpServers": { + "fabro": { + "command": "fabro", + "args": ["mcp", "start", "--server", "https://example.test/api/v1"] + } + } +} +``` + +`fabro mcp init ` writes the same entry into the agent config file under `mcpServers.fabro`, preserving every unrelated existing key. Re-running it is idempotent. If the existing file is invalid JSON or its root is not an object, fail clearly and do not overwrite it. + +Agent config paths: + +- `claude` + - macOS: `~/Library/Application Support/Claude/claude_desktop_config.json` + - Linux: `~/.config/Claude/claude_desktop_config.json` + - Windows: `%APPDATA%\Claude\claude_desktop_config.json` +- `cursor` + - all platforms: `~/.cursor/mcp.json` +- `windsurf` + - all platforms: `~/.codeium/windsurf/mcp_config.json` + +The MCP server exposes exactly these tools in this first slice: + +```text +fabro_run_create +fabro_run_search +fabro_run_interact +fabro_run_gather +fabro_run_events +``` + +### Tool Semantics + +`fabro_run_create` + +- Input: + +```rust +#[derive(Debug, Deserialize, JsonSchema)] +struct FabroRunCreateParams { + runs: Vec, +} + +#[derive(Debug, Deserialize, JsonSchema)] +struct CreateRunSpec { + workflow: String, + cwd: Option, + run_id: Option, + goal: Option, + #[serde(default)] + inputs: HashMap, + #[serde(default)] + labels: HashMap, + dry_run: Option, + auto_approve: Option, + model: Option, + provider: Option, + sandbox: Option, + preserve_sandbox: Option, + start: Option, +} +``` + +- `runs` is required and must contain 1 to 50 entries. +- `workflow` is a workflow path or project workflow selector resolved from `cwd` when provided, otherwise from the MCP process cwd. +- `start` defaults to `true` because this is analogous to Devin session creation: creating a run for an agent should normally launch it. Passing `start: false` creates a submitted run without starting it. +- `inputs` object values are converted to `toml::Value` with JSON-compatible semantics: string, bool, integer, float, arrays, and objects are accepted; null is rejected with a tool error naming the key. +- Output is structured: + +```rust +#[derive(Debug, Serialize, JsonSchema)] +struct CreateRunsResult { + runs: Vec, +} + +#[derive(Debug, Serialize, JsonSchema)] +struct CreatedRunResult { + run_id: String, + workflow: String, + started: bool, + status: String, +} +``` + +`fabro_run_search` + +- Input: + +```rust +struct FabroRunSearchParams { + run_ids: Option>, + workflow: Option, + labels: Option>, + status: Option>, + archived: Option, + created_after: Option, + created_before: Option, + first: Option, + after: Option, +} +``` + +- Search starts from `Client::list_store_runs()`, which already includes archived runs. +- `status` uses existing `run_status_kind(...)` strings. +- `created_after` and `created_before` parse RFC3339 timestamps or `YYYY-MM-DD` dates. +- `first` defaults to 20 and has max 100. +- `after` is an opaque cursor containing the last run id from the previous page. For the first implementation, encode it as the run id string and document it as opaque in the tool description. +- Output contains normalized run summaries: + +```rust +struct RunSummaryResult { + run_id: String, + workflow_name: String, + workflow_slug: Option, + status: String, + archived: bool, + created_at: String, + started_at: Option, + completed_at: Option, + labels: HashMap, + source_directory: Option, + repo_origin_url: Option, + goal: String, +} +``` + +`fabro_run_interact` + +- Input: + +```rust +#[derive(Debug, Deserialize, JsonSchema)] +#[serde(rename_all = "snake_case")] +enum RunInteractAction { + Get, + Start, + Message, + Cancel, + Archive, + Unarchive, + GetQuestions, + Answer, +} + +struct FabroRunInteractParams { + action: RunInteractAction, + run_id: String, + message: Option, + interrupt: Option, + question_id: Option, + answer: Option, +} +``` + +- `run_id` accepts the same selector semantics as CLI commands by calling `Client::resolve_run(...)`. +- `get` returns summary plus projection from `retrieve_run` and `get_run_state`. +- `start` calls `start_run(resume = false)`. +- `message` calls `steer_run`; `message` is required and trimmed; `interrupt` defaults false. +- `cancel` calls `cancel_run`. +- `archive` and `unarchive` call existing API methods. +- `get_questions` calls `list_run_questions`. +- `answer` requires `question_id` and maps answer JSON into `SubmitAnswerRequest`: + - boolean true -> yes + - boolean false -> no + - string -> freeform + - `{ "option": "key" }` -> single choice + - `{ "options": ["a", "b"] }` -> multi choice + - `{ "text": "..." }` -> freeform +- Return a structured object with `run_id`, `action`, and action-specific `result`. + +`fabro_run_gather` + +- Input: + +```rust +struct FabroRunGatherParams { + run_ids: Vec, + timeout_seconds: Option, + poll_interval_seconds: Option, +} +``` + +- `run_ids` is required, max 50. +- `timeout_seconds` defaults to 300 and maxes at 600. +- `poll_interval_seconds` defaults to 15 and mins at 5. +- Resolve selectors once at the start. +- Poll `retrieve_run` until every run is terminal or timeout expires. +- Output contains each final or current run summary plus `timed_out: bool`. + +`fabro_run_events` + +- Input: + +```rust +#[derive(Debug, Deserialize, JsonSchema)] +#[serde(rename_all = "snake_case")] +enum RunEventsAction { + List, + Details, + Search, +} + +struct FabroRunEventsParams { + action: RunEventsAction, + run_id: String, + event_types: Option>, + categories: Option>, + direction: Option, + created_after: Option, + created_before: Option, + first: Option, + after: Option, + event_ids: Option>, + offset: Option, + limit: Option, + max_content_length: Option, + query: Option, +} +``` + +- Use `Client::list_run_events(...)` rather than SSE for deterministic request/response behavior. +- `list` returns paginated envelopes sorted ascending by default; `direction: "desc"` reverses after fetching. +- `details` filters by `event_ids`. +- `search` filters events whose serialized event JSON contains `query`. +- `event_types` match `event.event_name()`. +- `categories` are best-effort derived from the prefix before the first `.` in `event_name`, for example `run.completed` has category `run`. +- `first` defaults to 50 and maxes at 200. `limit` is accepted as an alias for compatibility with the Devin-shaped input. `after` maps to `since_seq`. +- `max_content_length` defaults to 20_000 and truncates only large serialized event payload strings, with a `truncated: true` marker in the returned event item. + +### Contracts And Invariants + +- `fabro mcp start` stdout is reserved for MCP JSON-RPC only. All logs, warnings, errors, tracing, and diagnostics must go to stderr. +- MCP initialize and tools/list must not require a live Fabro server. API connection is lazy and happens when a tool needs it. +- There is no separate MCP authentication. Tool calls use the same CLI auth store and `fabro-client` behavior as existing CLI commands. Auth failures returned from tools must include the existing user guidance: `Run \`fabro auth login\` to authenticate.` +- Tool failures are MCP tool errors, not process exits. The stdio server should stay alive after invalid arguments, not-found selectors, conflicts, auth failures, and API errors. +- Tool-level argument validation that does not need server state must run before acquiring the lazy Fabro API client. Every handler must convert raw MCP parameter structs into its tool-specific `Validated...` type before auth lookup, client creation, selector resolution, or API calls. Invalid local input such as empty run lists, too many run ids, malformed timestamps, missing required action fields, unsupported answer JSON, or timeout values must report that validation error even when the CLI is not authenticated or the server is unavailable. +- Every successful tool returns structured content and a concise text fallback. The text fallback is for clients that do not yet show MCP structured output. Do not return `rmcp::Json` directly from successful tools, because its text content is the full JSON payload. Instead, build a `CallToolResult` with `structured_content: Some(...)` and a short `Content::text(...)` summary. +- `rmcp 1.3` only accepts manually constructed `CallToolResult` values from tool handlers through `Result`. Do not use `Result` in `#[tool]` methods; it does not satisfy `IntoCallToolResult`. Expected Fabro failures must be returned as `Ok(CallToolResult::error(...))` so they are MCP tool errors and the server stays alive. Reserve `Err(ErrorData)` for unexpected serialization/framework failures. +- Run selectors must go through `Client::resolve_run(...)` to preserve existing Fabro prefix/workflow-name behavior. +- Run creation must reuse `build_run_manifest(...)` and server manifest validation. Do not fabricate run specs or bypass the same source-of-truth path as `fabro create`. +- Agent config writes must be idempotent and preserve unrelated user config. +- Do not add live LLM/provider tests for this first slice. Use dry-run workflows and local/test servers. + +## Strategy Decisions + +- **Implement the server in `fabro-mcp-server`:** The existing `fabro-mcp` crate remains the client/shared protocol support for agents consuming third-party MCP servers. The new `fabro-mcp-server` crate owns the Fabro server implementation and exposes explicit settings APIs so `fabro-cli` can wire `fabro mcp ...` commands without making the existing client crate a server crate. +- **Use `rmcp` instead of hand-rolled JSON-RPC:** The project already depends on `rmcp` and uses it for MCP client behavior. The server should use the same SDK to get initialize/tools/list/tools/call semantics, JSON schema generation, and stdio framing right. +- **Default create to start:** Devin's session creation starts usable sessions. For Fabro, a run that stays submitted unless the caller remembers a second tool call is a surprising first-use experience. `start: false` keeps the lower-level control available without making it the default. +- **Use five Devin-shaped tools instead of many tiny tools:** The user explicitly asked to adapt Devin sessions to Fabro runs. The five-tool shape is easier for MCP clients to discover and keeps later additions compatible. Internally, the Rust implementation should still split actions into small functions. +- **Lazy API connection:** MCP clients often list tools during startup. Requiring auth/server connectivity during initialize would make even configuration validation brittle. Lazy connection gives users useful tool discovery and clear per-tool auth errors. +- **Validate before connecting:** MCP clients often probe tools with incomplete or malformed payloads. Local validation must happen before API client acquisition so callers get actionable schema/argument errors instead of misleading auth or server availability failures. + +## Task 1: Add CLI Surface And Help Snapshots + +**Files:** +- Create: `lib/crates/fabro-mcp-server/Cargo.toml` +- Create: `lib/crates/fabro-mcp-server/src/lib.rs` +- Create: `lib/crates/fabro-mcp-server/src/config.rs` +- Create: `lib/crates/fabro-mcp-server/src/run_tools.rs` +- Create: `lib/crates/fabro-mcp-server/src/server.rs` +- Modify: `lib/crates/fabro-cli/Cargo.toml` +- Modify: `lib/crates/fabro-cli/src/args.rs` +- Modify: `lib/crates/fabro-cli/src/main.rs` +- Modify: `lib/crates/fabro-cli/src/commands/mod.rs` +- Create: `lib/crates/fabro-cli/src/commands/mcp/mod.rs` +- Create: `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` +- Modify: `lib/crates/fabro-cli/tests/it/cmd/mod.rs` + +- [ ] **Step 1: Write failing CLI help tests** + +Add `mod mcp;` to `lib/crates/fabro-cli/tests/it/cmd/mod.rs`. + +Create `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` with snapshots for: + +```rust +use fabro_test::{fabro_snapshot, test_context}; + +#[test] +fn help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "--help"]); + fabro_snapshot!(context.filters(), cmd, @""); +} + +#[test] +fn start_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "start", "--help"]); + fabro_snapshot!(context.filters(), cmd, @""); +} + +#[test] +fn config_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "config", "--help"]); + fabro_snapshot!(context.filters(), cmd, @""); +} + +#[test] +fn init_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "init", "--help"]); + fabro_snapshot!(context.filters(), cmd, @""); +} +``` + +- [ ] **Step 2: Run the help tests and verify they fail** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::help cmd::mcp::start_help cmd::mcp::config_help cmd::mcp::init_help +``` + +Expected: FAIL because `fabro mcp` does not exist. + +- [ ] **Step 3: Add new crate, clap arguments, and no-op dispatch** + +Create `lib/crates/fabro-mcp-server/Cargo.toml` with the package name +`fabro-mcp-server`. Add direct `rmcp` dependency there: + +```toml +rmcp = { workspace = true, features = ["server", "macros", "schemars", "transport-io"] } +``` + +Also add the Fabro crate dependencies needed for settings/auth, API calls, +manifest construction, and tests. In `lib/crates/fabro-cli/Cargo.toml`, add only +the path dependency: + +```toml +fabro-mcp-server = { path = "../fabro-mcp-server" } +``` + +In `lib/crates/fabro-cli/src/args.rs`, add: + +```rust +#[derive(Args)] +pub(crate) struct McpNamespace { + #[command(subcommand)] + pub(crate) command: McpCommand, +} + +#[derive(Subcommand)] +pub(crate) enum McpCommand { + /// Start the Fabro MCP server over stdio + Start(McpStartArgs), + /// Print MCP client configuration JSON + Config(McpConfigArgs), + /// Configure an MCP client to launch Fabro + Init(McpInitArgs), +} + +#[derive(Args, Debug, Clone, Default)] +pub(crate) struct McpStartArgs { + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Args, Debug, Clone, Default)] +pub(crate) struct McpConfigArgs { + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Args, Debug, Clone)] +pub(crate) struct McpInitArgs { + pub(crate) agent: McpAgent, + + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Debug, Clone, Copy, ValueEnum)] +pub(crate) enum McpAgent { + Claude, + Cursor, + Windsurf, +} +``` + +Add `Commands::Mcp(McpNamespace)` with help text `Model Context Protocol server`. + +In `Commands::name()`: + +```rust +Self::Mcp(ns) => match &ns.command { + McpCommand::Start(_) => "mcp start", + McpCommand::Config(_) => "mcp config", + McpCommand::Init(_) => "mcp init", +}, +``` + +In `commands/mod.rs`, add `pub(crate) mod mcp;`. + +Create `commands/mcp/mod.rs`: + +```rust +use anyhow::Result; + +use crate::args::{McpAgent, McpCommand, McpNamespace, ServerConnectionArgs}; +use crate::command_context::CommandContext; + +pub(crate) async fn dispatch(ns: McpNamespace, base_ctx: &CommandContext) -> Result<()> { + match ns.command { + McpCommand::Start(args) => { + fabro_mcp_server::start(server_settings(base_ctx, &args.connection)?).await + } + McpCommand::Config(args) => { + let json = fabro_mcp_server::config_json(config_settings(&args.connection)?)?; + print!("{json}"); + Ok(()) + } + McpCommand::Init(args) => { + fabro_mcp_server::init_agent(init_settings(args.agent, &args.connection)?)?; + Ok(()) + } + } +} +``` + +Add small conversion helpers in `commands/mcp/mod.rs` that turn CLI arguments +and `base_ctx.cwd()` into `fabro_mcp_server` settings. These helpers must pass +plain owned values such as server URL override, storage-dir override, home dir, +and cwd; the new crate must not depend on `fabro-cli::CommandContext`. + +Create `lib/crates/fabro-mcp-server/src/lib.rs`, `config.rs`, `run_tools.rs`, +and `server.rs` in this task. Use stub implementations that return `Ok(())` or +placeholder JSON for config/init for now, except `start(settings)` can +`anyhow::bail!("fabro mcp start is not implemented yet")` until Task 3. +`run_tools.rs` can contain only a placeholder module comment until Task 3 adds +the first types/helpers. + +In `main.rs`, dispatch: + +```rust +Commands::Mcp(ns) => { + commands::mcp::dispatch(ns, &base_ctx).await?; +} +``` + +- [ ] **Step 4: Run help tests and accept expected snapshots** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::help cmd::mcp::start_help cmd::mcp::config_help cmd::mcp::init_help +cargo insta pending-snapshots +cargo insta accept +cargo nextest run -p fabro-cli --test it cmd::mcp::help cmd::mcp::start_help cmd::mcp::config_help cmd::mcp::init_help +``` + +Expected: first run produces snapshots to inspect, final run PASS. + +- [ ] **Step 5: Refactor and verify** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --all +cargo +nightly-2026-04-14 clippy -p fabro-cli --test it -- -D warnings +cargo +nightly-2026-04-14 clippy -p fabro-mcp-server --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add lib/crates/fabro-mcp-server/Cargo.toml lib/crates/fabro-mcp-server/src/lib.rs lib/crates/fabro-mcp-server/src/config.rs lib/crates/fabro-mcp-server/src/run_tools.rs lib/crates/fabro-mcp-server/src/server.rs lib/crates/fabro-cli/Cargo.toml lib/crates/fabro-cli/src/args.rs lib/crates/fabro-cli/src/main.rs lib/crates/fabro-cli/src/commands/mod.rs lib/crates/fabro-cli/src/commands/mcp/mod.rs lib/crates/fabro-cli/tests/it/cmd/mod.rs lib/crates/fabro-cli/tests/it/cmd/mcp.rs +git commit -m "feat(cli): add mcp command surface" +``` + +## Task 2: Implement `fabro mcp config` And `fabro mcp init` + +**Files:** +- Modify: `lib/crates/fabro-mcp-server/src/config.rs` +- Modify: `lib/crates/fabro-mcp-server/src/lib.rs` +- Modify: `lib/crates/fabro-cli/src/commands/mcp/mod.rs` +- Modify: `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` + +- [ ] **Step 1: Write failing config/init tests** + +Add tests: + +```rust +#[test] +fn config_prints_generic_mcp_json() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "config"]); + fabro_snapshot!(context.filters(), cmd, @""); +} + +#[test] +fn config_preserves_connection_flags() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args([ + "mcp", + "config", + "--server", + "https://example.test/api/v1", + "--storage-dir", + "/tmp/fabro-mcp-storage", + ]); + fabro_snapshot!(context.filters(), cmd, @""); +} + +#[test] +fn init_cursor_writes_idempotent_config() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "cursor"]) + .assert() + .success(); + context + .command() + .args(["mcp", "init", "cursor"]) + .assert() + .success(); + + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @""); +} + +#[test] +fn init_claude_writes_platform_config() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "claude"]) + .assert() + .success(); + + let config_path = expected_claude_config_path(&context.home_dir); + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @""); +} + +#[test] +fn init_windsurf_writes_config() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "windsurf"]) + .assert() + .success(); + + let config_path = context + .home_dir + .join(".codeium") + .join("windsurf") + .join("mcp_config.json"); + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @""); +} + +#[test] +fn init_preserves_existing_servers() { + let context = test_context!(); + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + std::fs::create_dir_all(config_path.parent().unwrap()).unwrap(); + std::fs::write( + &config_path, + r#"{"mcpServers":{"other":{"command":"other","args":["serve"]}},"theme":"dark"}"#, + ) + .unwrap(); + + context + .command() + .args(["mcp", "init", "cursor", "--server", "https://example.test/api/v1"]) + .assert() + .success(); + + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @""); +} + +#[test] +fn init_invalid_json_fails_without_overwrite() { + let context = test_context!(); + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + std::fs::create_dir_all(config_path.parent().unwrap()).unwrap(); + std::fs::write(&config_path, "{not json").unwrap(); + + let mut cmd = context.command(); + cmd.args(["mcp", "init", "cursor"]); + fabro_snapshot!(context.filters(), cmd, @""); + assert_eq!(std::fs::read_to_string(config_path).unwrap(), "{not json"); +} +``` + +Use `fabro_json_snapshot` where the parsed config is the contract. Add a small +`expected_claude_config_path(home_dir: &Path) -> PathBuf` helper in the test +module with the same platform branches as production so macOS, Linux, and +Windows path behavior is covered. Add the required import: + +```rust +use fabro_test::{fabro_json_snapshot, fabro_snapshot, test_context}; +``` + +- [ ] **Step 2: Run tests and verify they fail** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::config_prints_generic_mcp_json cmd::mcp::config_preserves_connection_flags cmd::mcp::init_cursor_writes_idempotent_config cmd::mcp::init_claude_writes_platform_config cmd::mcp::init_windsurf_writes_config cmd::mcp::init_preserves_existing_servers cmd::mcp::init_invalid_json_fails_without_overwrite +``` + +Expected: FAIL because config/init are stubs. + +- [ ] **Step 3: Implement config rendering** + +In `lib/crates/fabro-mcp-server/src/config.rs`, implement: + +```rust +#![expect( + clippy::disallowed_methods, + reason = "MCP client config setup intentionally performs small synchronous JSON file reads/writes from a CLI command." +)] + +use std::path::PathBuf; + +use anyhow::{Context as _, Result, anyhow, bail}; +use serde_json::{Map, Value, json}; + +const SERVER_NAME: &str = "fabro"; + +pub fn config_json(settings: McpConfigSettings) -> Result { + serde_json::to_string_pretty(&generic_config(&settings)) + .map(|json| format!("{json}\n")) + .context("failed to render Fabro MCP client config") +} + +pub fn init_agent(settings: McpInitSettings) -> Result<()> { + let path = agent_config_path(settings.agent, &settings.home_dir)?; + let entry = server_entry(&settings.config); + merge_server_entry(&path, entry)?; + Ok(()) +} +``` + +`server_entry(...)` must emit command `fabro` and args built by: + +```rust +fn start_args(settings: &McpConfigSettings) -> Vec { + let mut args = vec!["mcp".to_string(), "start".to_string()]; + if let Some(server) = settings.server.as_ref() { + args.push("--server".to_string()); + args.push(server.clone()); + } + if let Some(storage_dir) = settings.storage_dir.as_deref() { + args.push("--storage-dir".to_string()); + args.push(storage_dir.display().to_string()); + } + args +} +``` + +Implement `merge_server_entry(path, entry)` so it: + +- creates the parent directory +- reads existing JSON if the file exists +- rejects invalid JSON with context including the path +- rejects non-object roots and non-object `mcpServers` +- inserts/replaces only `mcpServers.fabro` +- writes pretty JSON plus trailing newline + +Implement all three supported path mappings (`claude`, `cursor`, `windsurf`). +For `claude`, use `dirs::home_dir()` plus platform cfgs: + +- macOS: `Library/Application Support/Claude/claude_desktop_config.json` +- Linux: `.config/Claude/claude_desktop_config.json` +- Windows: `%APPDATA%\Claude\claude_desktop_config.json`, falling back to `~/AppData/Roaming/Claude/claude_desktop_config.json` when `APPDATA` is absent. The fallback is needed because integration tests run the compiled binary under `fabro_test::apply_test_isolation`, which clears ambient `APPDATA`. + +- [ ] **Step 4: Run config/init tests and accept snapshots** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::config_prints_generic_mcp_json cmd::mcp::config_preserves_connection_flags cmd::mcp::init_cursor_writes_idempotent_config cmd::mcp::init_claude_writes_platform_config cmd::mcp::init_windsurf_writes_config cmd::mcp::init_preserves_existing_servers cmd::mcp::init_invalid_json_fails_without_overwrite +cargo insta pending-snapshots +cargo insta accept +cargo nextest run -p fabro-cli --test it cmd::mcp::config_prints_generic_mcp_json cmd::mcp::config_preserves_connection_flags cmd::mcp::init_cursor_writes_idempotent_config cmd::mcp::init_claude_writes_platform_config cmd::mcp::init_windsurf_writes_config cmd::mcp::init_preserves_existing_servers cmd::mcp::init_invalid_json_fails_without_overwrite +``` + +Expected: PASS. + +- [ ] **Step 5: Refactor and verify** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --all +cargo +nightly-2026-04-14 clippy -p fabro-cli --test it -- -D warnings +cargo +nightly-2026-04-14 clippy -p fabro-mcp-server --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 6: Commit** + +```bash +git add lib/crates/fabro-mcp-server/src/config.rs lib/crates/fabro-mcp-server/src/lib.rs lib/crates/fabro-cli/src/commands/mcp/mod.rs lib/crates/fabro-cli/tests/it/cmd/mcp.rs +git commit -m "feat(cli): configure fabro mcp clients" +``` + +## Task 3: Add MCP Server Skeleton With Protocol Tests + +**Files:** +- Modify: `lib/crates/fabro-mcp-server/src/server.rs` +- Modify: `lib/crates/fabro-mcp-server/src/run_tools.rs` +- Modify: `lib/crates/fabro-mcp-server/src/lib.rs` +- Modify: `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` + +- [ ] **Step 1: Write failing stdio protocol test** + +Add a test that uses the existing `fabro_mcp::client::McpClient` to spawn the compiled CLI: + +```rust +#[tokio::test(flavor = "multi_thread")] +async fn stdio_server_initializes_and_lists_run_tools() { + let context = test_context!(); + let config = fabro_mcp::config::McpServerSettings { + name: "fabro-under-test".to_string(), + transport: fabro_mcp::config::McpTransport::Stdio { + command: vec![ + env!("CARGO_BIN_EXE_fabro").to_string(), + "mcp".to_string(), + "start".to_string(), + ], + env: mcp_stdio_env(&context), + }, + startup_timeout_secs: 10, + tool_timeout_secs: 30, + }; + let client = fabro_mcp::client::McpClient::new(&config).unwrap(); + client.initialize(config.startup_timeout()).await.unwrap(); + + let tools = client.list_tools().await.unwrap(); + let names: Vec<_> = tools.iter().map(|(name, _, _)| name.as_str()).collect(); + assert_eq!( + names, + vec![ + "fabro_run_create", + "fabro_run_search", + "fabro_run_interact", + "fabro_run_gather", + "fabro_run_events", + ] + ); +} +``` + +`TestContext` does not currently expose a reusable command env map. Add a narrow +test helper that constructs a deterministic child-process environment instead +of reading from ambient user `HOME` or trying to reverse a built +`std::process::Command`: + +```rust +struct McpStdioFixture { + command: Vec, + env: HashMap, + current_dir: PathBuf, +} + +fn mcp_stdio_fixture(context: &fabro_test::TestContext, extra_args: &[&str]) -> McpStdioFixture { + let mut command = vec![ + env!("CARGO_BIN_EXE_fabro").to_string(), + "mcp".to_string(), + "start".to_string(), + ]; + command.extend(extra_args.iter().map(|arg| (*arg).to_string())); + + let mut env = fabro_test::isolated_env(&context.home_dir); + env.insert("HOME".to_string(), context.home_dir.display().to_string()); + env.insert("FABRO_HOME".to_string(), context.home_dir.join(".fabro").display().to_string()); + env.insert("NO_COLOR".to_string(), "1".to_string()); + + McpStdioFixture { + command, + env, + current_dir: context.temp_dir.clone(), + } +} +``` + +If `fabro_test::isolated_env` does not exist, add a similarly narrow helper to +`fabro_test` that returns the same env map used by +`fabro_test::apply_test_isolation`. Use `fixture.env.clone()` for +`fabro_mcp::config::McpServerSettings`, and use the same `fixture.env` plus +`fixture.current_dir` when spawning raw subprocess tests; raw subprocess helpers +must call `cmd.env_clear()` before applying this map. The production crate +should expose equivalent explicit settings (`McpServerSettings { server, +storage_dir, home_dir, cwd }`) so tests and CLI dispatch build settings from +owned values directly; do not depend on ambient process env in tests. + +- [ ] **Step 2: Run test and verify it fails** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::stdio_server_initializes_and_lists_run_tools +``` + +Expected: FAIL because `fabro mcp start` is not implemented. + +- [ ] **Step 3: Implement rmcp server skeleton** + +In `lib/crates/fabro-mcp-server/src/server.rs`, implement: + +```rust +use std::path::PathBuf; +use std::sync::Arc; + +use anyhow::Result; +use rmcp::{ + ErrorData, ServerHandler, serve_server, + handler::server::{router::tool::ToolRouter, wrapper::Parameters}, + model::{CallToolResult, ServerCapabilities, ServerInfo}, + tool, tool_handler, tool_router, + transport::stdio, +}; +use tokio::sync::OnceCell; + +use fabro_client::Client; + +use crate::{McpServerSettings, run_tools}; + +#[derive(Clone)] +pub(crate) struct FabroMcpServer { + settings: Arc, + client: Arc>>, + cwd: PathBuf, + tool_router: ToolRouter, +} + +pub async fn start(settings: McpServerSettings) -> Result<()> { + let server = FabroMcpServer::new(Arc::new(settings)); + let service = serve_server(server, stdio()).await?; + service.waiting().await?; + Ok(()) +} +``` + +Implement `ServerHandler` through the `#[tool_handler]` impl, not a separate +plain impl. `rmcp::serve_server(...)` returns after initialization with a +running service handle; `fabro mcp start` must await `service.waiting()` so the +stdio process stays alive for later `tools/list` and `tools/call` requests. + +```rust +#[tool_handler(router = self.tool_router)] +impl ServerHandler for FabroMcpServer { + fn get_info(&self) -> ServerInfo { + ServerInfo::new(ServerCapabilities::builder().enable_tools().build()) + .with_instructions("Use these tools to create, inspect, control, wait for, and read events from Fabro workflow runs.") + } +} +``` + +Add tool functions with temporary placeholder results: + +```rust +#[tool_router] +impl FabroMcpServer { + pub(crate) fn new(settings: Arc) -> Self { ... } + + #[tool(name = "fabro_run_create", description = "...")] + async fn fabro_run_create( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedCreateRuns::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::create_runs(client, &self.cwd, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::create_runs_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } +} + +``` + +Use this same handler shape for all five tools: first normalize and validate +the parameter object into a tool-specific `Validated...` type, then acquire the +lazy client only after validation succeeds, call the corresponding `run_tools` +function, return successful values with `success_result(...)`, and convert +expected Fabro/API/validation failures with `error_result(...)`. + +`new(...)` should copy `settings.cwd.clone()` into the `cwd` field before +storing the settings, so tool calls resolve relative workflows against the MCP +process cwd captured at startup. + +Each placeholder in `run_tools.rs` should still define the input structs, +validated parameter structs, and `TryFrom` validation hooks for all five +tools in this task. The run functions can return +`Err(ToolError::message("not implemented"))` until later tasks, except the +module must compile and tools must be listed. Add a small crate-local tool error +type: + +```rust +#[derive(Debug)] +pub(crate) struct ToolError { + message: String, +} + +impl ToolError { + pub(crate) fn message(message: impl Into) -> Self { + Self { + message: message.into(), + } + } + + pub(crate) fn from_anyhow(err: anyhow::Error) -> Self { + Self::message(format_tool_error(err)) + } + + pub(crate) fn as_str(&self) -> &str { + &self.message + } +} + +pub(crate) type ToolResult = Result; +``` + +Then add result helpers: + +```rust +pub(crate) fn success_result( + value: &T, + text: impl Into, +) -> Result { + let structured_content = serde_json::to_value(value).map_err(|err| { + rmcp::ErrorData::internal_error( + format!("failed to serialize Fabro MCP tool result: {err}"), + None, + ) + })?; + let mut result = rmcp::model::CallToolResult::structured(structured_content); + result.content = vec![rmcp::model::Content::text(text.into())]; + Ok(result) +} + +pub(crate) fn error_result(err: ToolError) -> rmcp::model::CallToolResult { + rmcp::model::CallToolResult::error(vec![rmcp::model::Content::text( + err.as_str().to_string(), + )]) +} +``` + +Add one text helper per result type, for example `create_runs_text(...)`, so +fallback content is concise: `"created 1 Fabro run and started 1"`, not a full +JSON dump. + +Implement `client(&self)` with lazy connection: + +```rust +async fn client(&self) -> Result, run_tools::ToolError> { + self.client + .get_or_try_init(|| async { client_from_settings(&self.settings).await.map_err(run_tools::ToolError::from_anyhow) }) + .await + .map(Arc::clone) +} +``` + +Make `format_tool_error` append auth guidance when `fabro_util::exit::exit_class_for(&err) == Some(ExitClass::AuthRequired)`. + +- [ ] **Step 4: Run protocol test** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::stdio_server_initializes_and_lists_run_tools +``` + +Expected: PASS listing all five tools. + +- [ ] **Step 5: Add stdout-purity regression** + +Add a raw subprocess test that: + +- spawns `fabro mcp start` +- sends a JSON-RPC initialize request on stdin +- reads the first stdout line +- asserts it parses as JSON and has `jsonrpc: "2.0"` +- asserts stderr may contain logs but stdout contains no leading human text + +Use `mcp_stdio_fixture(&context, &[])` for the raw subprocess helper so this +test and the `McpServerSettings` test use identical command, env, and cwd +values. + +Use a child timeout and kill-on-drop cleanup. The raw JSON should be: + +```json +{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"fabro-test","version":"0.0.0"}}} +``` + +- [ ] **Step 6: Run protocol checks** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::stdio_server_initializes_and_lists_run_tools cmd::mcp::stdio_start_writes_only_json_rpc_to_stdout +``` + +Expected: PASS. + +- [ ] **Step 7: Add startup/list-tools performance smoke** + +Add a lightweight smoke test that uses `mcp_stdio_fixture` to initialize the +server and call `tools/list` without a live Fabro server or auth. Assert the +combined initialize plus list-tools path completes within 2 seconds on the test +machine: + +```rust +#[tokio::test(flavor = "multi_thread")] +async fn stdio_startup_and_list_tools_is_fast() { + let context = test_context!(); + let start = std::time::Instant::now(); + let client = spawn_mcp_client(&context, &[]).await; + let tools = client.list_tools().await.unwrap(); + assert_eq!(tools.len(), 5); + assert!(start.elapsed() < std::time::Duration::from_secs(2)); +} +``` + +This is a smoke check, not a benchmark. If CI variance makes 2 seconds too +tight, keep the assertion but adjust the threshold in the implementation with a +comment explaining the observed bound. + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::stdio_startup_and_list_tools_is_fast +``` + +Expected: PASS. + +- [ ] **Step 8: Refactor and verify** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --all +cargo +nightly-2026-04-14 clippy -p fabro-cli --test it -- -D warnings +cargo +nightly-2026-04-14 clippy -p fabro-mcp-server --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 9: Commit** + +```bash +git add lib/crates/fabro-mcp-server/src/server.rs lib/crates/fabro-mcp-server/src/run_tools.rs lib/crates/fabro-mcp-server/src/lib.rs lib/crates/fabro-cli/tests/it/cmd/mcp.rs +git commit -m "feat(cli): start fabro mcp stdio server" +``` + +## Task 4: Implement Run Create/Search Tools + +**Files:** +- Modify: `lib/crates/fabro-mcp-server/src/run_tools.rs` +- Modify: `lib/crates/fabro-cli/src/commands/run/overrides.rs` +- Modify: `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` + +- [ ] **Step 1: Write failing create/search integration test** + +Add a test backed by an authenticated real Fabro server: + +```rust +#[tokio::test(flavor = "multi_thread")] +async fn mcp_create_and_search_manage_real_runs_with_cli_auth() { + let context = test_context!(); + let harness = RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let workflow = context.install_fixture("simple.fabro"); + + let client = spawn_mcp_client(&context, &[ + "--server", + &target_url, + ]).await; + + let create = call_tool_json(&client, "fabro_run_create", serde_json::json!({ + "runs": [{ + "workflow": workflow, + "dry_run": true, + "auto_approve": true, + "labels": { "source": "mcp-test" } + }] + })).await; + let run_id = create["runs"][0]["run_id"].as_str().unwrap().to_string(); + assert_eq!(create["runs"][0]["started"], true); + + let search = call_tool_json(&client, "fabro_run_search", serde_json::json!({ + "run_ids": [run_id], + "labels": { "source": "mcp-test" }, + "first": 10 + })).await; + fabro_json_snapshot!(context, normalize_run_search(search), @""); + + harness.shutdown().await; +} +``` + +Implement `call_tool_json(...)` so it asserts `is_error != Some(true)`, extracts +`structured_content`, and verifies the first text content is present and does +not start with `{` or `[`; this makes the concise fallback contract automated +instead of a manual-only check. + +If `RealAuthHarness::start_with_dev_token` cannot create runs due missing server settings for local execution, use `TestContext` managed server plus explicit `seed_dev_token_auth` against its server target, but keep the test proving persisted CLI auth is used. + +- [ ] **Step 2: Run test and verify it fails** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::mcp_create_and_search_manage_real_runs_with_cli_auth +``` + +Expected: FAIL because tools return not implemented. + +- [ ] **Step 3: Implement shared parameter and result types** + +In `run_tools.rs`, define public crate-visible structs for every tool input/output with: + +```rust +#[derive(Debug, serde::Deserialize, rmcp::schemars::JsonSchema)] +pub(crate) struct ... + +#[derive(Debug, serde::Serialize, rmcp::schemars::JsonSchema)] +pub(crate) struct ... +``` + +Use `#[serde(default)]` on optional map fields so omitted maps become empty maps where helpful. + +- [ ] **Step 4: Expose manifest override helpers** + +In `commands/run/overrides.rs`, make these helpers `pub(crate)` if needed: + +- `parse_labels` +- `model_from_args` +- `sandbox_layer` +- `execution_layer` +- `goal_layer_from_args` + +If changing visibility creates awkward API, instead add one new crate-visible function: + +```rust +pub(crate) fn manifest_overrides_from_parts(input: ManifestOverrideParts<'_>) -> Result +``` + +Prefer the single helper if more than three helpers would need visibility changes. + +- [ ] **Step 5: Implement `fabro_run_create`** + +Implementation outline: + +```rust +pub(crate) async fn create_runs( + client: Arc, + base_cwd: &Path, + params: ValidatedCreateRuns, +) -> ToolResult { + let mut created = Vec::with_capacity(params.runs.len()); + for spec in params.runs { + let cwd = spec.cwd.clone().unwrap_or_else(|| base_cwd.to_path_buf()); + let run_id = spec.run_id.as_deref().map(str::parse).transpose().map_err(tool_err)?; + let overrides = build_mcp_manifest_overrides(&spec, &cwd)?; + let manifest_args = mcp_manifest_args(&spec); + let built = build_run_manifest(ManifestBuildInput { + workflow: PathBuf::from(&spec.workflow), + cwd, + run_overrides: overrides.run, + cli_overrides: overrides.cli, + input_overrides: overrides.input_overrides, + args: manifest_args, + run_id, + user_settings_path: Some(active_settings_path(None)), + })?; + let validation = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?; + reject_validation_errors(validation)?; + let run_id = client.create_run_from_manifest(built.manifest).await?; + let started = spec.start.unwrap_or(true); + if started { + client.start_run(&run_id, false).await?; + } + let summary = client.retrieve_run(&run_id).await?; + created.push(CreatedRunResult::from_summary(summary, started)); + } + Ok(CreateRunsResult { runs: created }) +} +``` + +Important: the function signature in `server.rs` should pass both the lazy API client and the MCP process cwd from the captured `McpServerSettings`, not call `std::env::current_dir()` deep in the tool. The raw `FabroRunCreateParams` must only appear at the MCP handler boundary; `ValidatedCreateRuns::try_from(raw)` must run before acquiring the client, and `create_runs(...)` must accept `ValidatedCreateRuns`. +The outline above omits some `map_err(...)` calls for readability; the real +implementation must convert every `anyhow::Error` and API error into +`ToolError` with the shared formatting helper so `?` never tries to convert +`anyhow::Error` directly into `ToolError`. + +Implement `mcp_manifest_args(&CreateRunSpec) -> Option` +in `run_tools.rs`. It should mirror `manifest_builder::run_manifest_args` for +provenance: + +```rust +fn mcp_manifest_args(spec: &CreateRunSpec) -> Option { + let label = spec + .labels + .iter() + .map(|(key, value)| format!("{key}={value}")) + .collect::>(); + let input = spec + .inputs + .iter() + .map(|(key, value)| format!("{key}={value}")) + .collect::>(); + let payload = types::ManifestArgs { + auto_approve: spec.auto_approve.filter(|value| *value), + dry_run: spec.dry_run.filter(|value| *value), + label, + model: spec.model.clone(), + preserve_sandbox: spec.preserve_sandbox.filter(|value| *value), + provider: spec.provider.clone(), + sandbox: spec.sandbox.clone(), + docker_image: None, + input, + verbose: None, + }; + (!mcp_manifest_args_is_empty(&payload)).then_some(payload) +} +``` + +Keep the emptiness check local if `manifest_args_is_empty` is not accessible. +The `input` strings are only for provenance; authoritative input values come +from `input_overrides` after JSON-to-TOML conversion. + +- [ ] **Step 6: Implement JSON-to-TOML input conversion** + +Add unit tests in `run_tools.rs` for: + +- strings +- bools +- integers +- floats +- arrays +- objects +- null rejected with key name + +Run: + +```bash +cargo nextest run -p fabro-mcp-server run_tools +``` + +Expected: PASS after implementation. + +- [ ] **Step 7: Implement `fabro_run_search`** + +Use existing `server_runs::ServerRunSummaryInfo` where useful, but avoid adding public API only for tests. Search should: + +- fetch `client.list_store_runs().await` +- sort newest first using created/start timestamp and run id as tie-breaker +- apply filters +- page with `first` and `after` +- return `SearchRunsResult { runs, next_cursor }` + +Do not drop archived runs by default. `archived: Some(false)` should exclude them. + +- [ ] **Step 8: Run create/search tests** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::mcp_create_and_search_manage_real_runs_with_cli_auth +``` + +Expected: PASS. + +- [ ] **Step 9: Refactor and verify** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --all +cargo +nightly-2026-04-14 clippy -p fabro-cli --test it -- -D warnings +cargo +nightly-2026-04-14 clippy -p fabro-mcp-server --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 10: Commit** + +```bash +git add lib/crates/fabro-mcp-server/src/run_tools.rs lib/crates/fabro-cli/src/commands/run/overrides.rs lib/crates/fabro-cli/tests/it/cmd/mcp.rs +git commit -m "feat(cli): add mcp run create and search tools" +``` + +## Task 5: Implement Interact/Gather/Events Tools + +**Files:** +- Modify: `lib/crates/fabro-mcp-server/src/run_tools.rs` +- Modify: `lib/crates/fabro-cli/tests/it/cmd/mcp.rs` + +- [ ] **Step 1: Write failing lifecycle interaction test** + +Add a test that: + +- creates a dry-run auto-approved run with `fabro_run_create` +- calls `fabro_run_gather` with the run id +- calls `fabro_run_interact` action `get` +- calls `fabro_run_events` action `list` +- calls `fabro_run_interact` action `archive` +- calls `fabro_run_interact` action `unarchive` +- verifies server-visible state through API or a follow-up `fabro_run_search` + +Snapshot a normalized object: + +```rust +fabro_json_snapshot!( + context, + serde_json::json!({ + "gather": normalize_gather(gather), + "get_status": get["result"]["summary"]["status"], + "events_nonempty": events["events"].as_array().unwrap().is_empty() == false, + "archive_action": archive["action"], + "unarchive_action": unarchive["action"], + }), + @"" +); +``` + +- [ ] **Step 2: Write failing validation/error tests** + +Add tests for: + +- `fabro_run_gather` rejects more than 50 run ids without requiring auth or a reachable server. Start `fabro mcp start --server http://127.0.0.1:9` with no auth entry, call the tool, assert the error mentions `run_ids`, then call `tools/list` again to prove the server stayed alive. +- `fabro_run_gather` returns `timed_out: true` when the timeout expires before all requested runs are terminal. Use a real authenticated test server, create or select a non-terminal run, call gather with `timeout_seconds: 1` and `poll_interval_seconds: 5`, assert elapsed wall time is bounded, and verify the returned run summary is the current state rather than a process/tool failure. +- `fabro_run_interact` action `message` without `message` returns an MCP tool error and the server remains alive for a subsequent `fabro_run_search`. +- Missing auth against a protected remote target returns a tool error containing `Run \`fabro auth login\` to authenticate.` + +- [ ] **Step 3: Run tests and verify they fail** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::mcp_lifecycle_tools_manage_real_run cmd::mcp::mcp_gather_rejects_too_many_runs cmd::mcp::mcp_gather_returns_timeout_result cmd::mcp::mcp_interact_error_does_not_stop_server cmd::mcp::mcp_tool_auth_error_mentions_login +``` + +Expected: FAIL because tools are incomplete. + +- [ ] **Step 4: Implement `fabro_run_interact`** + +Implement one small function per action: + +```rust +async fn interact_get(client: &Client, run_id: &RunId) -> Result +async fn interact_start(client: &Client, run_id: &RunId) -> Result +async fn interact_message(client: &Client, run_id: &RunId, message: Option, interrupt: bool) -> Result +async fn interact_cancel(client: &Client, run_id: &RunId) -> Result +async fn interact_archive(client: &Client, run_id: &RunId) -> Result +async fn interact_unarchive(client: &Client, run_id: &RunId) -> Result +async fn interact_get_questions(client: &Client, run_id: &RunId) -> Result +async fn interact_answer(client: &Client, run_id: &RunId, question_id: Option, answer: Option) -> Result +``` + +Resolve selectors once: + +```rust +let run_id = client.resolve_run(¶ms.run_id).await?.id; +``` + +Use `serde_json::to_value(...)` for API objects rather than manually copying complex projection/question structures. + +- [ ] **Step 5: Implement answer mapping tests and helper** + +Add unit tests for: + +```rust +assert_answer_json(json!(true), json!({"kind": "yes"})); +assert_answer_json(json!(false), json!({"kind": "no"})); +assert_answer_json(json!("hello"), json!({"kind": "text", "text": "hello"})); +assert_answer_json(json!({"option":"a"}), json!({"kind": "selected", "option_key": "a"})); +assert_answer_json( + json!({"options":["a","b"]}), + json!({"kind": "multi_selected", "option_keys": ["a", "b"]}), +); +assert_answer_json(json!({"text":"hello"}), json!({"kind": "text", "text": "hello"})); +``` + +Build the generated `fabro_api::types::SubmitAnswerRequest` through the +documented wire JSON shape and then serialize it back in tests: + +```rust +fn answer_to_submit_request(answer: serde_json::Value) -> ToolResult { + let payload = match answer { + serde_json::Value::Bool(true) => serde_json::json!({ "kind": "yes" }), + serde_json::Value::Bool(false) => serde_json::json!({ "kind": "no" }), + serde_json::Value::String(text) => serde_json::json!({ "kind": "text", "text": text }), + serde_json::Value::Object(mut object) => { + if let Some(option) = object.remove("option") { + serde_json::json!({ "kind": "selected", "option_key": option }) + } else if let Some(options) = object.remove("options") { + serde_json::json!({ "kind": "multi_selected", "option_keys": options }) + } else if let Some(text) = object.remove("text") { + serde_json::json!({ "kind": "text", "text": text }) + } else { + return Err(ToolError::message( + "answer object must contain one of: option, options, text", + )); + } + } + other => { + return Err(ToolError::message(format!( + "unsupported answer value: {other}; expected boolean, string, or object", + ))); + } + }; + serde_json::from_value(payload).map_err(|err| { + ToolError::message(format!("failed to build submit-answer request: {err}")) + }) +} +``` + +This matches the current API contract proven by +`lib/crates/fabro-api/tests/submit_answer_request_round_trip.rs`, which uses +`kind: yes`, `kind: no`, `kind: selected`, `kind: multi_selected`, and +`kind: text`. Do not introduce references to non-existent generated names such +as `SubmitAnswerRequestKind`. + +- [ ] **Step 6: Implement `fabro_run_gather`** + +Validation converts raw `FabroRunGatherParams` into `ValidatedGatherRuns` +before client acquisition: + +```rust +validate_len("run_ids", params.run_ids.len(), 1, 50)?; +let timeout = params.timeout_seconds.unwrap_or(300).min(600); +let poll = params.poll_interval_seconds.unwrap_or(15).max(5); +``` + +Implementation: + +- resolve all selectors at the start +- poll summaries until every `summary.lifecycle.status.is_terminal()` or deadline +- if the deadline expires, return `timed_out: true`, current run summaries, and `elapsed_seconds` as a successful structured tool result +- only return a tool error for selector/API failures, not for ordinary timeout expiry + +- [ ] **Step 7: Implement `fabro_run_events`** + +Fetch events using: + +```rust +let events = client + .list_run_events(&run_id, params.after, effective_limit_for_fetch(params)) + .await?; +``` + +Then apply filters in memory: + +- event ids +- event types using `event.event.event_name()` +- categories +- created_after/before +- query substring on serialized event JSON +- offset +- limit/first +- direction +- max_content_length truncation + +Output: + +```rust +struct RunEventsResult { + run_id: String, + action: RunEventsAction, + events: Vec, + next_cursor: Option, +} +``` + +`next_cursor` is last returned sequence plus one when at least one event was returned. + +- [ ] **Step 8: Run lifecycle and error tests** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp::mcp_lifecycle_tools_manage_real_run cmd::mcp::mcp_gather_rejects_too_many_runs cmd::mcp::mcp_gather_returns_timeout_result cmd::mcp::mcp_interact_error_does_not_stop_server cmd::mcp::mcp_tool_auth_error_mentions_login +``` + +Expected: PASS. + +- [ ] **Step 9: Refactor and verify** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --all +cargo +nightly-2026-04-14 clippy -p fabro-cli --test it -- -D warnings +cargo +nightly-2026-04-14 clippy -p fabro-mcp-server --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 10: Commit** + +```bash +git add lib/crates/fabro-mcp-server/src/run_tools.rs lib/crates/fabro-cli/tests/it/cmd/mcp.rs +git commit -m "feat(cli): add mcp run control tools" +``` + +## Task 6: Final Contract Coverage And Workspace Verification + +**Files:** +- Modify as needed from prior tasks only. + +- [ ] **Step 1: Run all MCP command tests** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it cmd::mcp +``` + +Expected: PASS. + +- [ ] **Step 2: Run existing relevant MCP client tests** + +Run: + +```bash +cargo nextest run -p fabro-mcp +``` + +Expected: PASS. This confirms the existing external MCP client crate was not regressed by dependency feature unification. + +- [ ] **Step 3: Run relevant existing CLI run/auth tests** + +Run: + +```bash +cargo nextest run -p fabro-cli --test it scenario::auth::auth_login_refresh_logout_flow scenario::lifecycle::dry_run_create_start_attach_works_with_default_run_lookup cmd::ps::ps_explicit_local_tcp_target_uses_auth_store +``` + +Expected: PASS. If exact test names drift, use `cargo nextest list -p fabro-cli --test it | rg 'auth_login_refresh_logout_flow|dry_run_create_start_attach|explicit_local_tcp'` and run the matching tests. + +- [ ] **Step 4: Run formatting and linting** + +Run: + +```bash +cargo +nightly-2026-04-14 fmt --check --all +cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings +``` + +Expected: PASS. + +- [ ] **Step 5: Run broader regression suite** + +Run: + +```bash +ulimit -n 4096 && cargo nextest run --workspace +``` + +Expected: PASS. + +- [ ] **Step 6: Inspect snapshots before accepting any remaining changes** + +Run: + +```bash +cargo insta pending-snapshots +``` + +Expected: no pending snapshots. If pending snapshots exist, inspect them before accepting. Only accept snapshots caused by this feature. + +- [ ] **Step 7: Final code review pass** + +Check manually: + +- `fabro mcp start` has no `printout!`, `println!`, `eprintln!` is only for stderr and not in server steady-state startup. +- all MCP tool argument validation returns tool errors, not process exits. +- successful MCP tools include both `structuredContent` and short text content; text content is not just serialized JSON. +- no tests write run internals directly. +- no live provider credentials are required. +- agent config merge preserves unrelated keys. +- auth failures include login guidance. + +- [ ] **Step 8: Commit final fixes if any** + +```bash +git status --short +git add +git commit -m "test(cli): cover fabro mcp server contract" +``` + +Only make this commit if Task 6 produced additional fixes or tests not already committed. diff --git a/docs/plans/2026-05-11-daytona-real-agent-smoke-qa-plan.md b/docs/plans/2026-05-11-daytona-real-agent-smoke-qa-plan.md new file mode 100644 index 000000000..21a6d3516 --- /dev/null +++ b/docs/plans/2026-05-11-daytona-real-agent-smoke-qa-plan.md @@ -0,0 +1,762 @@ +# Sandbox Real-Agent Smoke QA Plan + +## Purpose + +Manually smoke test the `add-acp-backend` branch with real LLM-backed agents across the full sandbox/agent matrix: + +| Sandbox provider | Claude | Codex | Gemini | +| --- | --- | --- | --- | +| Local | Required | Required | Required | +| Docker | Required | Required | Required | +| Daytona | Required | Required | Required | + +The key branch constraint is intentional: ACP requires bidirectional raw stdio and is supported by local and Docker in this cutover, but not by Daytona. This QA plan proves ACP works through local and Docker sandboxes for Claude, Codex, and Gemini; uses Daytona CLI execution for the same three agents as the positive Daytona coverage; and finally verifies ACP-on-Daytona fails clearly instead of falling back to host execution or PTY transport. + +## Scope + +In scope: + +- A real Claude ACP-backed agent running through the local sandbox provider with no container. +- A real Codex ACP-backed agent running through the local sandbox provider with no container. +- A real Gemini ACP-backed agent running through the local sandbox provider with no container. +- A real Claude ACP-backed agent running in a Docker sandbox. +- A real Codex ACP-backed agent running in a Docker sandbox. +- A real Gemini ACP-backed agent running in a Docker sandbox. +- A real Claude CLI-backed agent running in a Daytona sandbox. +- A real Codex CLI-backed agent running in a Daytona sandbox. +- A real Gemini CLI-backed agent running in a Daytona sandbox. +- Optional Daytona API backend control coverage after the required 3x3 matrix. +- An ACP-backed node on Daytona returning the expected unsupported-provider failure. +- Evidence capture through `inspect`, `events`, `dump`, and optional preserved-sandbox SSH. + +Out of scope: + +- Automated nextest coverage. +- ACP positive execution on Daytona. +- Full regression of Docker or local ACP behavior beyond this real-agent smoke. +- Snapshot creation performance tuning beyond what is needed to run the smoke. + +## Preconditions + +- Current branch is `add-acp-backend`. +- Local host has Node/npx available for the no-container ACP smoke. +- Docker is available for the Docker sandbox smoke. +- Daytona API key is available with sandbox/snapshot scopes. +- Real LLM credentials are available for all three required agents. +- GitHub access is configured if the operator chooses not to use `skip_clone = true`. +- Network access from the host, Docker container, and Daytona sandbox allows installing CLI packages. + +Recommended environment: + +```bash +cargo build -p fabro-cli +export FABRO=./target/debug/fabro + +set -a +source .env +set +a + +$FABRO doctor -v +``` + +Required environment variables for the full matrix: + +- `DAYTONA_API_KEY` +- `ANTHROPIC_API_KEY` +- `OPENAI_API_KEY` +- `GEMINI_API_KEY` + +## Test Data Setup + +Create a scratch directory for manual smoke files: + +```bash +mkdir -p smoke tmp +``` + +Docker and Daytona smoke configs use `skip_clone = true` to avoid depending on pushed branch state. This keeps the test focused on runtime behavior and real agent execution. The local smoke intentionally runs directly in the current working tree because `provider = "local"` has no container boundary; remove `smoke_local_acp_*_result.txt`, `.fabro-smoke-*-acp`, and `.fabro-smoke-home/` during cleanup if they remain. + +Agent definitions for the required matrix: + +| Agent | Provider | Model | Credential | ACP command | Daytona CLI install | +| --- | --- | --- | --- | --- | --- | +| Claude | `anthropic` | `claude-haiku-4-5` | `ANTHROPIC_API_KEY` | `npx -y @zed-industries/claude-code-acp@latest` | `npm install -g @anthropic-ai/claude-code`; binary: `claude` | +| Codex | `openai` | `gpt-5.3-codex` | `OPENAI_API_KEY` | `npx -y @zed-industries/codex-acp@latest` | `npm install -g @openai/codex`; binary: `codex` | +| Gemini | `gemini` | `gemini-3.1-pro-preview` | `GEMINI_API_KEY` | `npx -y -- @google/gemini-cli@latest --experimental-acp` | `npm install -g @google/gemini-cli`; binary: `gemini` | + +## Smoke 1: Local ACP Backend Matrix + +### Goal + +Prove real Claude, Codex, and Gemini ACP-backed agents can run through the local sandbox provider without a container, use bidirectional raw stdio, and mutate the local workflow filesystem. + +Required local combinations: + +| Agent | Workflow | Result file | Expected content | +| --- | --- | --- | --- | +| Claude | `smoke/local_acp_claude.toml` | `smoke_local_acp_claude_result.txt` | `local-acp-claude-ok` | +| Codex | `smoke/local_acp_codex.toml` | `smoke_local_acp_codex_result.txt` | `local-acp-codex-ok` | +| Gemini | `smoke/local_acp_gemini.toml` | `smoke_local_acp_gemini_result.txt` | `local-acp-gemini-ok` | + +### Files + +Create one graph per agent. The Claude graph is: + +```dot +digraph LocalAcpClaudeSmoke { + graph [goal="Local ACP Claude backend smoke"] + start [shape=Mdiamond] + setup [shape=parallelogram, script="rm -f smoke_local_acp_claude_result.txt"] + work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="/bin/bash .fabro-smoke-claude-acp", prompt="Create a file named smoke_local_acp_claude_result.txt containing exactly: local-acp-claude-ok"] + verify [shape=parallelogram, script="test \"$(cat smoke_local_acp_claude_result.txt)\" = \"local-acp-claude-ok\" && cat smoke_local_acp_claude_result.txt"] + exit [shape=Msquare] + start -> setup -> work -> verify -> exit +} +``` + +Create matching Codex and Gemini graphs with these substitutions: + +| Agent | Graph file | Provider | Model | ACP wrapper | Result file | Expected content | +| --- | --- | --- | --- | --- | --- | --- | +| Codex | `smoke/local_acp_codex.fabro` | `openai` | `gpt-5.3-codex` | `.fabro-smoke-codex-acp` | `smoke_local_acp_codex_result.txt` | `local-acp-codex-ok` | +| Gemini | `smoke/local_acp_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `.fabro-smoke-gemini-acp` | `smoke_local_acp_gemini_result.txt` | `local-acp-gemini-ok` | + +Create `smoke/local_acp_claude.toml`: + +```toml +_version = 1 + +[workflow] +graph = "local_acp_claude.fabro" + +[run.sandbox] +provider = "local" + +[[run.prepare.steps]] +script = ''' +set -eu +NODE_DIR="$(dirname "$(command -v node)")" +NPX_PATH="$(command -v npx)" +cat > .fabro-smoke-claude-acp <-ok` content. +- `fabro events --tail 200` for each run includes: + - `agent.acp.started` + - `agent.acp.completed` + - `stage.completed` for `work` + - `stage.completed` for `verify` +- Events for each run do not include `agent.session.activated` for the `work` stage. +- Events for each run do not include `agent.cli.started` for the `work` stage. +- `fabro inspect ` shows each run succeeded. +- The local working tree contains each expected `smoke_local_acp__result.txt` file with exactly the expected content. + +### Failure Notes + +- Missing `node` or `npx` on the host is a setup failure for this no-container smoke. +- A successful run with API or CLI events instead of ACP events is a branch failure because ACP silently fell back to another backend. +- The local provider writes directly into the current working tree; check `git status` before cleanup. + +## Smoke 2: Docker ACP Backend Matrix + +### Goal + +Prove real Claude, Codex, and Gemini ACP-backed agents can run inside Docker sandboxes, use bidirectional non-PTY stdio through Docker exec, and mutate only the container workspace. + +Required Docker combinations: + +| Agent | Workflow | Result file | Expected content | +| --- | --- | --- | --- | +| Claude | `smoke/docker_acp_claude.toml` | `smoke_docker_acp_claude_result.txt` | `docker-acp-claude-ok` | +| Codex | `smoke/docker_acp_codex.toml` | `smoke_docker_acp_codex_result.txt` | `docker-acp-codex-ok` | +| Gemini | `smoke/docker_acp_gemini.toml` | `smoke_docker_acp_gemini_result.txt` | `docker-acp-gemini-ok` | + +### Files + +Create one graph per agent. The Claude graph is: + +```dot +digraph DockerAcpClaudeSmoke { + graph [goal="Docker ACP Claude backend smoke"] + start [shape=Mdiamond] + setup [shape=parallelogram, script="rm -f smoke_docker_acp_claude_result.txt"] + work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="/bin/bash .fabro-smoke-claude-acp", prompt="Create a file named smoke_docker_acp_claude_result.txt containing exactly: docker-acp-claude-ok"] + verify [shape=parallelogram, script="test \"$(cat smoke_docker_acp_claude_result.txt)\" = \"docker-acp-claude-ok\" && cat smoke_docker_acp_claude_result.txt"] + exit [shape=Msquare] + start -> setup -> work -> verify -> exit +} +``` + +Create matching Codex and Gemini graphs with these substitutions: + +| Agent | Graph file | Provider | Model | ACP wrapper | Result file | Expected content | +| --- | --- | --- | --- | --- | --- | --- | +| Codex | `smoke/docker_acp_codex.fabro` | `openai` | `gpt-5.3-codex` | `.fabro-smoke-codex-acp` | `smoke_docker_acp_codex_result.txt` | `docker-acp-codex-ok` | +| Gemini | `smoke/docker_acp_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `.fabro-smoke-gemini-acp` | `smoke_docker_acp_gemini_result.txt` | `docker-acp-gemini-ok` | + +Create `smoke/docker_acp_claude.toml`: + +```toml +_version = 1 + +[workflow] +graph = "docker_acp_claude.fabro" + +[run.sandbox] +provider = "docker" +preserve = true + +[run.sandbox.docker] +image = "buildpack-deps:noble" +network_mode = "bridge" +memory_limit = "4GB" +cpu_quota = 200000 +skip_clone = true + +[[run.prepare.steps]] +script = ''' +set -eu +mkdir -p "$HOME/.local" +if ! command -v node >/dev/null 2>&1; then + curl -fsSL https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.gz | tar -xz --strip-components=1 -C "$HOME/.local" +fi +export PATH="$HOME/.local/bin:$PATH" +npm --version +npx --version +NODE_DIR="$(dirname "$(command -v node)")" +NPX_PATH="$(command -v npx)" +cat > .fabro-smoke-claude-acp <-ok` content. +- `fabro events --tail 200` for each run includes: + - `sandbox.ready` + - `setup.started` + - `setup.completed` + - `agent.acp.started` + - `agent.acp.completed` + - `stage.completed` for `verify` +- Events for each run do not include `agent.session.activated` for the `work` stage. +- Events for each run do not include `agent.cli.started` for the `work` stage. +- `fabro inspect ` shows each run succeeded. +- Each preserved Docker sandbox contains its expected `smoke_docker_acp__result.txt` file with exactly the expected content. + +### Failure Notes + +- Docker daemon, image pull, or package-install failures are setup failures unless the error indicates ACP stdio or sandbox routing broke. +- A successful run with API or CLI events instead of ACP events is a branch failure because ACP silently fell back to another backend. + +## Smoke 3: Daytona API Backend Control + +### Goal + +Optionally prove a real provider API agent can use Fabro-managed tools inside the Daytona sandbox and mutate the sandbox filesystem. This is a backend control smoke, not part of the required 3x3 external-agent matrix. + +### Files + +Create `smoke/daytona_api.fabro`: + +```dot +digraph DaytonaApiSmoke { + graph [goal="Daytona API backend smoke"] + start [shape=Mdiamond] + setup [shape=parallelogram, script="rm -f smoke_api_result.txt"] + work [type="agent", backend="api", provider="anthropic", model="claude-haiku-4-5", prompt="Create a file named smoke_api_result.txt containing exactly: daytona-api-ok"] + verify [shape=parallelogram, script="test \"$(cat smoke_api_result.txt)\" = \"daytona-api-ok\" && cat smoke_api_result.txt"] + exit [shape=Msquare] + start -> setup -> work -> verify -> exit +} +``` + +Create `smoke/daytona_api.toml`: + +```toml +_version = 1 + +[workflow] +graph = "daytona_api.fabro" + +[run.sandbox] +provider = "daytona" +preserve = true + +[run.sandbox.daytona] +skip_clone = true +auto_stop_interval = 60 +``` + +### Run + +```bash +$FABRO run --auto-approve smoke/daytona_api.toml +``` + +### Pass Criteria + +- Run exits successfully. +- The `verify` stage prints `daytona-api-ok`. +- `fabro events --tail 200` includes: + - `sandbox.ready` + - `agent.session.activated` + - `stage.completed` for `work` + - `stage.completed` for `verify` +- `fabro inspect ` shows the run succeeded. + +### Failure Notes + +- Provider authentication failures are setup failures unless the error indicates sandbox routing or missing Daytona state. +- Missing `smoke_api_result.txt` after a successful agent stage is a failure. + +## Smoke 4: Daytona CLI Backend Matrix + +### Goal + +Prove the branch runs real Claude, Codex, and Gemini external CLI agents inside Daytona when the CLI is preinstalled by the workflow environment. This also confirms Fabro no longer installs CLIs implicitly at stage runtime. + +Required Daytona combinations: + +| Agent | Workflow | Result file | Expected content | CLI binary | +| --- | --- | --- | --- | --- | +| Claude | `smoke/daytona_cli_claude.toml` | `smoke_daytona_cli_claude_result.txt` | `daytona-cli-claude-ok` | `claude` | +| Codex | `smoke/daytona_cli_codex.toml` | `smoke_daytona_cli_codex_result.txt` | `daytona-cli-codex-ok` | `codex` | +| Gemini | `smoke/daytona_cli_gemini.toml` | `smoke_daytona_cli_gemini_result.txt` | `daytona-cli-gemini-ok` | `gemini` | + +### Files + +Create one graph per agent. The Claude graph is: + +```dot +digraph DaytonaCliClaudeSmoke { + graph [goal="Daytona CLI Claude backend smoke"] + start [shape=Mdiamond] + setup [shape=parallelogram, script="rm -f smoke_daytona_cli_claude_result.txt"] + work [type="agent", backend="cli", provider="anthropic", model="claude-haiku-4-5", prompt="Create a file named smoke_daytona_cli_claude_result.txt containing exactly: daytona-cli-claude-ok"] + verify [shape=parallelogram, script="test \"$(cat smoke_daytona_cli_claude_result.txt)\" = \"daytona-cli-claude-ok\" && cat smoke_daytona_cli_claude_result.txt"] + exit [shape=Msquare] + start -> setup -> work -> verify -> exit +} +``` + +Create matching Codex and Gemini graphs with these substitutions: + +| Agent | Graph file | Provider | Model | Result file | Expected content | +| --- | --- | --- | --- | --- | --- | +| Codex | `smoke/daytona_cli_codex.fabro` | `openai` | `gpt-5.3-codex` | `smoke_daytona_cli_codex_result.txt` | `daytona-cli-codex-ok` | +| Gemini | `smoke/daytona_cli_gemini.fabro` | `gemini` | `gemini-3.1-pro-preview` | `smoke_daytona_cli_gemini_result.txt` | `daytona-cli-gemini-ok` | + +Create `smoke/daytona_cli_claude.toml`: + +```toml +_version = 1 + +[workflow] +graph = "daytona_cli_claude.fabro" + +[run.sandbox] +provider = "daytona" +preserve = true + +[run.sandbox.daytona] +skip_clone = true +auto_stop_interval = 60 + +[[run.prepare.steps]] +script = ''' +set -eu +mkdir -p "$HOME/.local" +if ! command -v node >/dev/null 2>&1; then + curl -fsSL https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.gz | tar -xz --strip-components=1 -C "$HOME/.local" +fi +export PATH="$HOME/.local/bin:$PATH" +npm config set prefix "$HOME/.local" +command -v claude >/dev/null 2>&1 || npm install -g @anthropic-ai/claude-code +claude --version +''' +``` + +Create matching Codex and Gemini TOML files: + +- `smoke/daytona_cli_codex.toml` uses `graph = "daytona_cli_codex.fabro"`, installs `@openai/codex` when `codex` is missing, and prints `codex --version`. +- `smoke/daytona_cli_gemini.toml` uses `graph = "daytona_cli_gemini.fabro"`, installs `@google/gemini-cli` when `gemini` is missing, and prints `gemini --version`. + +### Run + +```bash +$FABRO run --auto-approve smoke/daytona_cli_claude.toml +$FABRO run --auto-approve smoke/daytona_cli_codex.toml +$FABRO run --auto-approve smoke/daytona_cli_gemini.toml +``` + +### Pass Criteria + +- Each of the three Daytona CLI runs exits successfully. +- Each `verify` stage prints its expected `daytona-cli--ok` content. +- `fabro events --tail 200` for each run includes: + - `setup.started` + - `setup.completed` + - `agent.cli.started` + - `agent.cli.completed` + - `stage.completed` for `verify` +- Events for each run do not include new `cli.ensure.started`, `cli.ensure.completed`, or `cli.ensure.failed` entries for this branch's runtime path. +- Each preserved Daytona sandbox contains its expected `smoke_daytona_cli__result.txt` file with exactly the expected content. + +### Failure Notes + +- `CLI backend requires '' to be installed in the sandbox PATH` means the prepare step did not install the agent CLI where the backend expects it. Treat this as environment/setup failure unless the prepare logs prove `claude`, `codex`, or `gemini` was installed in `$HOME/.local/bin`. +- CLI package-install failures may be caused by Daytona network policy or npm registry availability. + +## Smoke 5: Daytona ACP Backend Expected Unsupported Failure + +### Goal + +Prove ACP on Daytona fails explicitly because Daytona lacks bidirectional raw stdio support. This provider-boundary check is run once with Claude because the failure must happen before any agent-specific command is launched. This test guards against unsafe fallbacks such as running ACP on the host or over a PTY. + +### Files + +Create `smoke/daytona_acp_unsupported.fabro`: + +```dot +digraph DaytonaAcpUnsupportedSmoke { + graph [goal="ACP Daytona unsupported smoke"] + start [shape=Mdiamond] + work [type="agent", backend="acp", provider="anthropic", model="claude-haiku-4-5", acp_command="npx -y @zed-industries/claude-code-acp@latest", prompt="Create smoke_acp_result.txt"] + exit [shape=Msquare] + start -> work -> exit +} +``` + +Create `smoke/daytona_acp_unsupported.toml`: + +```toml +_version = 1 + +[workflow] +graph = "daytona_acp_unsupported.fabro" + +[run.sandbox] +provider = "daytona" +preserve = true + +[run.sandbox.daytona] +skip_clone = true +auto_stop_interval = 60 +``` + +### Run + +```bash +$FABRO run --auto-approve smoke/daytona_acp_unsupported.toml +``` + +### Pass Criteria + +- Run fails. +- The failure text contains: + - `ACP backend requires bidirectional stdio` + - `Daytona sandbox provider does not support it yet` +- Events include `agent.acp.started`. +- Events do not include `agent.acp.completed`. +- The preserved Daytona sandbox does not contain `smoke_acp_result.txt`. + +### Failure Notes + +- If the run succeeds, that is a failure for this branch because ACP should not execute on Daytona. +- If the failure is about `acp_command` missing, the workflow file is wrong. +- If the failure is about `npx` missing before the Daytona unsupported error, inspect the code path: the smoke should prove the sandbox stdio provider boundary, not package availability. + +## Required 3x3 Matrix Record + +The plan is incomplete until all nine sandbox/agent combinations below have a run ID and evidence: + +| Sandbox | Agent | Backend | Workflow | Run ID | +| --- | --- | --- | --- | --- | +| Local | Claude | ACP | `smoke/local_acp_claude.toml` | `________________` | +| Local | Codex | ACP | `smoke/local_acp_codex.toml` | `________________` | +| Local | Gemini | ACP | `smoke/local_acp_gemini.toml` | `________________` | +| Docker | Claude | ACP | `smoke/docker_acp_claude.toml` | `________________` | +| Docker | Codex | ACP | `smoke/docker_acp_codex.toml` | `________________` | +| Docker | Gemini | ACP | `smoke/docker_acp_gemini.toml` | `________________` | +| Daytona | Claude | CLI | `smoke/daytona_cli_claude.toml` | `________________` | +| Daytona | Codex | CLI | `smoke/daytona_cli_codex.toml` | `________________` | +| Daytona | Gemini | CLI | `smoke/daytona_cli_gemini.toml` | `________________` | + +## Verification Checklist + +Use this checklist as the operator-facing record for the smoke. Fill in run IDs and notes as each step completes. + +### Preconditions + +- [ ] Current branch is `add-acp-backend`. +- [ ] `cargo build -p fabro-cli` completed successfully. +- [ ] `FABRO=./target/debug/fabro` is exported for the shell running the smoke. +- [ ] `.env` is loaded. +- [ ] `$FABRO doctor -v` completed without a blocking environment error. +- [ ] Host `node` is available for the local ACP smoke. +- [ ] Host `npx` is available for the local ACP smoke. +- [ ] Docker is available for the Docker sandbox smoke. +- [ ] `DAYTONA_API_KEY` is present and has sandbox/snapshot scopes. +- [ ] `ANTHROPIC_API_KEY` is present for Claude smokes. +- [ ] `OPENAI_API_KEY` is present for Codex smokes. +- [ ] `GEMINI_API_KEY` is present for Gemini smokes. +- [ ] Host, Docker, and Daytona network paths can install CLI packages. +- [ ] `smoke/` and `tmp/` directories exist. + +### Smoke 1: Local ACP Backend Matrix + +- [ ] Created `smoke/local_acp_claude.fabro` and `smoke/local_acp_claude.toml`. +- [ ] Created `smoke/local_acp_codex.fabro` and `smoke/local_acp_codex.toml`. +- [ ] Created `smoke/local_acp_gemini.fabro` and `smoke/local_acp_gemini.toml`. +- [ ] All three local configs use `provider = "local"`. +- [ ] Prepare steps create `.fabro-smoke-claude-acp`, `.fabro-smoke-codex-acp`, and `.fabro-smoke-gemini-acp`. +- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_claude.toml`. +- [ ] Recorded local Claude ACP run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_codex.toml`. +- [ ] Recorded local Codex ACP run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/local_acp_gemini.toml`. +- [ ] Recorded local Gemini ACP run ID: `________________`. +- [ ] Each local ACP run exited successfully. +- [ ] Verify stages printed `local-acp-claude-ok`, `local-acp-codex-ok`, and `local-acp-gemini-ok`. +- [ ] Each local run's events include `agent.acp.started`. +- [ ] Each local run's events include `agent.acp.completed`. +- [ ] Each local run's events include `stage.completed` for `work`. +- [ ] Each local run's events include `stage.completed` for `verify`. +- [ ] Each local run's events do not include `agent.session.activated` for the `work` stage. +- [ ] Each local run's events do not include `agent.cli.started` for the `work` stage. +- [ ] `fabro inspect ` shows each local run succeeded. +- [ ] Local working tree contains the three expected `smoke_local_acp__result.txt` files with exact contents. + +### Smoke 2: Docker ACP Backend Matrix + +- [ ] Created `smoke/docker_acp_claude.fabro` and `smoke/docker_acp_claude.toml`. +- [ ] Created `smoke/docker_acp_codex.fabro` and `smoke/docker_acp_codex.toml`. +- [ ] Created `smoke/docker_acp_gemini.fabro` and `smoke/docker_acp_gemini.toml`. +- [ ] All three Docker configs use Docker with `preserve = true`. +- [ ] All three Docker configs use `skip_clone = true`. +- [ ] Prepare steps install or verify Node. +- [ ] Prepare steps verify `npx`. +- [ ] Prepare steps create `.fabro-smoke-claude-acp`, `.fabro-smoke-codex-acp`, and `.fabro-smoke-gemini-acp`. +- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_claude.toml`. +- [ ] Recorded Docker Claude ACP run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_codex.toml`. +- [ ] Recorded Docker Codex ACP run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/docker_acp_gemini.toml`. +- [ ] Recorded Docker Gemini ACP run ID: `________________`. +- [ ] Each Docker ACP run exited successfully. +- [ ] Verify stages printed `docker-acp-claude-ok`, `docker-acp-codex-ok`, and `docker-acp-gemini-ok`. +- [ ] Each Docker run's events include `sandbox.ready`. +- [ ] Each Docker run's events include `setup.started`. +- [ ] Each Docker run's events include `setup.completed`. +- [ ] Each Docker run's events include `agent.acp.started`. +- [ ] Each Docker run's events include `agent.acp.completed`. +- [ ] Each Docker run's events include `stage.completed` for `verify`. +- [ ] Each Docker run's events do not include `agent.session.activated` for the `work` stage. +- [ ] Each Docker run's events do not include `agent.cli.started` for the `work` stage. +- [ ] `fabro inspect ` shows each Docker run succeeded. +- [ ] Preserved Docker sandboxes contain their expected `smoke_docker_acp__result.txt` files with exact contents. + +### Smoke 3: Daytona API Backend Control + +- [ ] Created `smoke/daytona_api.fabro`. +- [ ] Created `smoke/daytona_api.toml`. +- [ ] Config uses Daytona with `preserve = true`. +- [ ] Config uses `skip_clone = true`. +- [ ] Ran `$FABRO run --auto-approve smoke/daytona_api.toml`. +- [ ] Recorded Daytona API smoke run ID: `________________`. +- [ ] Run exited successfully. +- [ ] `verify` stage printed `daytona-api-ok`. +- [ ] Events include `sandbox.ready`. +- [ ] Events include `agent.session.activated`. +- [ ] Events include `stage.completed` for `work`. +- [ ] Events include `stage.completed` for `verify`. +- [ ] `fabro inspect ` shows the run succeeded. +- [ ] Preserved Daytona sandbox contains `smoke_api_result.txt` with exactly `daytona-api-ok`. + +### Smoke 4: Daytona CLI Backend Matrix + +- [ ] Created `smoke/daytona_cli_claude.fabro` and `smoke/daytona_cli_claude.toml`. +- [ ] Created `smoke/daytona_cli_codex.fabro` and `smoke/daytona_cli_codex.toml`. +- [ ] Created `smoke/daytona_cli_gemini.fabro` and `smoke/daytona_cli_gemini.toml`. +- [ ] All three Daytona CLI configs use Daytona with `preserve = true`. +- [ ] All three Daytona CLI configs use `skip_clone = true`. +- [ ] Prepare steps install or verify Node. +- [ ] Prepare steps install or verify `claude`, `codex`, and `gemini` in the sandbox PATH. +- [ ] Prepare steps print `claude --version`, `codex --version`, and `gemini --version`. +- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_claude.toml`. +- [ ] Recorded Daytona Claude CLI run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_codex.toml`. +- [ ] Recorded Daytona Codex CLI run ID: `________________`. +- [ ] Ran `$FABRO run --auto-approve smoke/daytona_cli_gemini.toml`. +- [ ] Recorded Daytona Gemini CLI run ID: `________________`. +- [ ] Each Daytona CLI run exited successfully. +- [ ] Verify stages printed `daytona-cli-claude-ok`, `daytona-cli-codex-ok`, and `daytona-cli-gemini-ok`. +- [ ] Each Daytona CLI run's events include `setup.started`. +- [ ] Each Daytona CLI run's events include `setup.completed`. +- [ ] Each Daytona CLI run's events include `agent.cli.started`. +- [ ] Each Daytona CLI run's events include `agent.cli.completed`. +- [ ] Each Daytona CLI run's events include `stage.completed` for `verify`. +- [ ] Each Daytona CLI run's events do not include `cli.ensure.started`. +- [ ] Each Daytona CLI run's events do not include `cli.ensure.completed`. +- [ ] Each Daytona CLI run's events do not include `cli.ensure.failed`. +- [ ] Preserved Daytona sandboxes contain their expected `smoke_daytona_cli__result.txt` files with exact contents. + +### Smoke 5: Daytona ACP Unsupported Failure + +- [ ] Created `smoke/daytona_acp_unsupported.fabro`. +- [ ] Created `smoke/daytona_acp_unsupported.toml`. +- [ ] Config uses Daytona with `preserve = true`. +- [ ] Config uses `skip_clone = true`. +- [ ] Ran `$FABRO run --auto-approve smoke/daytona_acp_unsupported.toml`. +- [ ] Recorded Daytona ACP smoke run ID: `________________`. +- [ ] Run failed. +- [ ] Failure text contains `ACP backend requires bidirectional stdio`. +- [ ] Failure text contains `Daytona sandbox provider does not support it yet`. +- [ ] Events include `agent.acp.started`. +- [ ] Events do not include `agent.acp.completed`. +- [ ] Preserved Daytona sandbox does not contain `smoke_acp_result.txt`. +- [ ] No evidence shows ACP ran on the host. +- [ ] No evidence shows ACP used a PTY fallback. +- [ ] No evidence shows ACP silently fell back to API or CLI. + +### Evidence Capture + +For each required run: + +- [ ] Captured `$FABRO inspect `. +- [ ] Captured `$FABRO events --tail 200`. +- [ ] Captured `$FABRO dump --output tmp/-dump `. +- [ ] Recorded command used. +- [ ] Recorded final status. +- [ ] Recorded relevant event names. +- [ ] Recorded any external-provider or sandbox infrastructure errors. + +For provider-specific filesystem checks: + +- [ ] Inspected local working tree files directly. +- [ ] Inspected preserved Docker filesystem with `$FABRO sandbox ssh `. +- [ ] Inspected preserved Daytona filesystem with `$FABRO sandbox ssh `. +- [ ] Recorded whether expected files exist in the expected provider workspace. + +### Cleanup And Final Acceptance + +- [ ] Removed each run with `$FABRO rm -f ` after evidence capture. +- [ ] Removed local smoke artifacts: `smoke_local_acp_*_result.txt`, `.fabro-smoke-*-acp`, and `.fabro-smoke-home/`. +- [ ] Verified preserved Docker sandbox is gone with Docker or `fabro inspect `. +- [ ] Verified preserved Daytona sandboxes are gone from the Daytona dashboard or `fabro inspect `. +- [ ] Local ACP backend smoke succeeded with Claude, Codex, and Gemini and no API/CLI fallback. +- [ ] Docker ACP backend smoke succeeded with Claude, Codex, and Gemini and no API/CLI fallback. +- [ ] Daytona CLI backend smoke succeeded with Claude, Codex, and Gemini after explicit CLI installation in prepare steps. +- [ ] Optional Daytona API backend control result was recorded if run. +- [ ] Daytona ACP smoke failed with the expected unsupported bidirectional-stdio message. +- [ ] Captured events and dumps are sufficient to diagnose any failure without rerunning immediately. + +## Evidence Capture Commands + +For each run, capture: + +```bash +$FABRO inspect +$FABRO events --tail 200 +$FABRO dump --output tmp/-dump +``` + +For the local smoke, inspect the local filesystem. For preserved Docker and Daytona sandboxes, inspect the provider filesystem: + +```bash +cat smoke_local_acp_claude_result.txt 2>/dev/null || true +cat smoke_local_acp_codex_result.txt 2>/dev/null || true +cat smoke_local_acp_gemini_result.txt 2>/dev/null || true + +$FABRO sandbox ssh +pwd +ls -la +cat smoke_docker_acp_claude_result.txt 2>/dev/null || true +cat smoke_docker_acp_codex_result.txt 2>/dev/null || true +cat smoke_docker_acp_gemini_result.txt 2>/dev/null || true +cat smoke_api_result.txt 2>/dev/null || true +cat smoke_daytona_cli_claude_result.txt 2>/dev/null || true +cat smoke_daytona_cli_codex_result.txt 2>/dev/null || true +cat smoke_daytona_cli_gemini_result.txt 2>/dev/null || true +cat smoke_acp_result.txt 2>/dev/null || true +exit +``` + +Record for each run: + +- Run ID. +- Command used. +- Final status. +- Relevant event names. +- Whether expected files exist in the expected local, Docker, or Daytona workspace. +- Any external-provider or sandbox infrastructure errors. + +## Cleanup + +After evidence capture: + +```bash +$FABRO rm -f +rm -f smoke_local_acp_*_result.txt .fabro-smoke-*-acp +rm -rf .fabro-smoke-home +``` + +Verify preserved Docker and Daytona sandboxes are gone from Docker/Daytona or by rerunning `fabro inspect ` and confirming no active sandbox remains. + +## Final Acceptance Criteria + +The branch passes this manual QA plan when: + +1. The local sandbox provider succeeds with real Claude, Codex, and Gemini ACP-backed agents. +2. The Docker sandbox provider succeeds with real Claude, Codex, and Gemini ACP-backed agents. +3. The Daytona sandbox provider succeeds with real Claude, Codex, and Gemini CLI-backed agents after explicit CLI installation in prepare steps. +4. The optional Daytona API control smoke, if run, succeeds or has a clearly recorded external-provider/setup failure. +5. The ACP Daytona smoke fails with the expected unsupported bidirectional-stdio message. +6. No evidence shows ACP-on-Daytona ran on the host, used a PTY fallback, or silently fell back to API/CLI. +7. Captured run events and dumps are sufficient to diagnose any failure without rerunning immediately. diff --git a/docs/public/agents/mcp.mdx b/docs/public/agents/mcp.mdx index f6d74d057..06a5ce4da 100644 --- a/docs/public/agents/mcp.mdx +++ b/docs/public/agents/mcp.mdx @@ -1,11 +1,40 @@ --- title: "MCP" -description: "Extend agents with Model Context Protocol servers" +description: "Connect MCP tools to agents and expose Fabro runs to MCP clients" --- MCP ([Model Context Protocol](https://modelcontextprotocol.io/)) lets you connect external tool servers to Fabro agents. An MCP server exposes tools over a standardized protocol — databases, APIs, file systems, custom services — and Fabro discovers and registers them automatically. Agents call MCP tools the same way they call built-in tools. -## How it works +Fabro can also run as an MCP server. MCP clients can use Fabro's run-management tools to create, inspect, control, wait for, and read events from workflow runs through the authenticated `fabro` CLI. + +## Fabro as an MCP server + +Use `fabro mcp init` to configure an MCP client to launch Fabro: + +```bash +fabro mcp init claude +``` + +Supported client targets are `claude`, `cursor`, and `windsurf`. The generated configuration launches `fabro mcp start` over stdio and reuses the CLI's normal server selection, OAuth refresh, dev-token handling, proxy behavior, and local storage. + +You can also print the MCP configuration JSON or start the server directly: + +```bash +fabro mcp config +fabro mcp start +``` + +Pass `--server` when the MCP client should connect to a specific Fabro server, or `--storage-dir` when it should use a non-default CLI storage directory. + +| Tool | Purpose | +|---|---| +| `fabro_run_create` | Create one or more workflow runs, starting them by default. | +| `fabro_run_search` | Search runs by ID, workflow, labels, status, archive state, and creation time. | +| `fabro_run_interact` | Get, start, message, cancel, archive, unarchive, inspect questions, or answer a run. | +| `fabro_run_gather` | Wait for runs to reach terminal states, returning current state on timeout. | +| `fabro_run_events` | List, inspect, or search stored events for a run. | + +## Fabro agents as MCP clients When an agent session starts with MCP servers configured, Fabro: @@ -26,12 +55,12 @@ mcp__{server}__{tool} For example, a server named `filesystem` exposing a `read_file` tool becomes `mcp__filesystem__read_file`. Special characters in server or tool names (hyphens, dots, etc.) are replaced with underscores. -## Configuration +## Agent MCP configuration -MCP servers can be configured in two places: +MCP servers available to Fabro agents can be configured in two places: -- **`~/.fabro/settings.toml`** — applies to `fabro exec` sessions. See [User Configuration](/reference/user-configuration#mcp_servers-section). -- **Run config TOML** — applies to workflow runs (`fabro run`). See [Run Configuration](/execution/run-configuration#mcp_servers). +- **User configuration** — `~/.fabro/settings.toml` can define shared workflow MCP servers under `[run.agent.mcps.]`, or `fabro exec`-only servers under `[cli.exec.agent.mcps.]`. See [User Configuration](/reference/user-configuration). +- **Run config TOML** — workflow config can define run-specific MCP servers under `[run.agent.mcps.]`. See [Run Configuration](/execution/run-configuration#runagentmcps). Each server entry specifies a transport type and optional timeouts. The server name is the TOML table key and is used in qualified tool names. @@ -42,13 +71,13 @@ Each server entry specifies a transport type and optional timeouts. The server n The most common transport. Fabro spawns a child process on the host and communicates over stdin/stdout: ```toml -[mcp_servers.filesystem] +[run.agent.mcps.filesystem] type = "stdio" command = ["npx", "-y", "@modelcontextprotocol/server-filesystem", "/workspace"] -startup_timeout_secs = 15 -tool_timeout_secs = 90 +startup_timeout = "15s" +tool_timeout = "90s" -[mcp_servers.filesystem.env] +[run.agent.mcps.filesystem.env] NODE_ENV = "production" ``` @@ -56,20 +85,21 @@ NODE_ENV = "production" |---|---|---| | `type` | Must be `"stdio"`. | — | | `command` | Array: the executable followed by its arguments. | — | +| `script` | Shell script alternative to `command`. | — | | `env` | Additional environment variables for the child process. | `{}` | -| `startup_timeout_secs` | Max seconds to wait for the MCP handshake. | `10` | -| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` | +| `startup_timeout` | Max duration to wait for the MCP handshake. | `"10s"` | +| `tool_timeout` | Max duration for a single tool call. | `"60s"` | ### HTTP For remote MCP servers accessible over Streamable HTTP: ```toml -[mcp_servers.sentry] +[run.agent.mcps.sentry] type = "http" url = "https://mcp.sentry.dev/mcp" -[mcp_servers.sentry.headers] +[run.agent.mcps.sentry.headers] Authorization = "Bearer sk-xxx" ``` @@ -78,30 +108,31 @@ Authorization = "Bearer sk-xxx" | `type` | Must be `"http"`. | — | | `url` | The MCP server endpoint URL. | — | | `headers` | Optional HTTP headers (e.g., for authentication). | `{}` | -| `startup_timeout_secs` | Max seconds to wait for the MCP handshake. | `10` | -| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` | +| `startup_timeout` | Max duration to wait for the MCP handshake. | `"10s"` | +| `tool_timeout` | Max duration for a single tool call. | `"60s"` | ### Sandbox Runs the MCP server inside the workflow's sandbox (e.g., a [Daytona](/integrations/daytona) cloud VM). Fabro starts the server as a background process, waits for it to listen on the specified port, obtains an authenticated preview URL, and connects via HTTP. This is the right transport for MCP servers that need access to the sandbox environment — for example, [Playwright](https://github.com/microsoft/playwright-mcp) for browser automation. ```toml -[mcp_servers.playwright] +[run.agent.mcps.playwright] type = "sandbox" command = ["npx", "@playwright/mcp@latest", "--port", "3100", "--headless", "--browser", "chromium"] port = 3100 -startup_timeout_secs = 60 -tool_timeout_secs = 120 +startup_timeout = "60s" +tool_timeout = "2m" ``` | Field | Description | Default | |---|---|---| | `type` | Must be `"sandbox"`. | — | | `command` | Array: the command to run inside the sandbox. Must include a flag that makes the server listen on `port`. | — | +| `script` | Shell script alternative to `command`. | — | | `port` | The port the MCP server listens on inside the sandbox. | — | | `env` | Additional environment variables for the server process. | `{}` | -| `startup_timeout_secs` | Max seconds to wait for the server to start listening and complete the MCP handshake. | `10` | -| `tool_timeout_secs` | Max seconds to wait for a single tool call. | `60` | +| `startup_timeout` | Max duration to wait for the server to start listening and complete the MCP handshake. | `"10s"` | +| `tool_timeout` | Max duration for a single tool call. | `"60s"` | The sandbox transport requires a remote sandbox provider (Daytona) that supports preview URLs. During session initialization, Fabro: @@ -117,7 +148,7 @@ The sandbox transport requires a remote sandbox provider (Daytona) that supports Each MCP server is started sequentially during session initialization. The startup sequence for each server is: 1. **Spawn / connect** — For stdio, spawn the child process. For HTTP, create the HTTP client. For sandbox, start the server inside the sandbox and connect via preview URL. -2. **Handshake** — Perform the MCP protocol handshake within the `startup_timeout_secs` window. +2. **Handshake** — Perform the MCP protocol handshake within the `startup_timeout` window. 3. **Tool discovery** — Call `tools/list` to enumerate available tools. 4. **Registration** — Add each tool to the agent's registry with its qualified name. @@ -132,7 +163,7 @@ When the LLM calls an MCP tool: 3. The server executes the tool and returns a result 4. The result is converted to text and returned to the LLM as a tool result -Tool calls are subject to the `tool_timeout_secs` configured on the server. If a call exceeds the timeout, it fails with a timeout error. +Tool calls are subject to the `tool_timeout` configured on the server. If a call exceeds the timeout, it fails with a timeout error. ### Content handling @@ -197,7 +228,7 @@ MCP servers that fail to start do not block the agent session. The agent proceed ## Protocol details -Fabro implements the MCP client side using the `rmcp` SDK (protocol version `2025-03-26`). The client identifies itself as `fabro-mcp` and supports: +For agent-side MCP connections, Fabro implements the MCP client side using the `rmcp` SDK (protocol version `2025-03-26`). The client identifies itself as `fabro-mcp` and supports: - Tool listing and invocation - Server logging notifications (routed to Fabro's tracing system) diff --git a/docs/public/agents/outputs.mdx b/docs/public/agents/outputs.mdx index 2fda1a5a8..1407f4db6 100644 --- a/docs/public/agents/outputs.mdx +++ b/docs/public/agents/outputs.mdx @@ -127,7 +127,7 @@ The tracked paths are stored as `files_touched` on the stage outcome: For the **API backend**, Fabro subscribes to agent session events. When a `ToolCallStarted` event fires for `write_file` or `edit_file`, Fabro records the `file_path` argument as pending. When the corresponding `ToolCallCompleted` arrives without an error, the path is confirmed as touched. Failed tool calls are discarded. -For the **CLI backend**, Fabro takes a different approach: it runs `git diff --name-only` and `git ls-files --others --exclude-standard` before and after the agent session, then computes the difference. Any files that appear in the "after" snapshot but not "before" are recorded as touched. +For the **CLI** and **ACP** backends, Fabro takes a different approach: it runs `git diff --name-only` and `git ls-files --others --exclude-standard` before and after the external agent session, then computes the difference. Any files that appear in the "after" snapshot but not "before" are recorded as touched. ## Artifact offloading diff --git a/docs/public/agents/subagents.mdx b/docs/public/agents/subagents.mdx index cc3741ed1..0427b9fa8 100644 --- a/docs/public/agents/subagents.mdx +++ b/docs/public/agents/subagents.mdx @@ -6,7 +6,7 @@ description: "Delegate subtasks to child agent sessions" An agent can spawn **sub-agents** to delegate work to independent child sessions. Each sub-agent gets its own LLM session and tool access, runs concurrently with the parent, and returns its result when finished. -Sub-agents are only available with the [API backend](/core-concepts/agents#api-backend-default) (the default). Agents using the [CLI backend](/core-concepts/agents#cli-backend) cannot spawn sub-agents. +Sub-agents are only available with the [API backend](/core-concepts/agents#api-backend-default) (the default). Agents using the [CLI backend](/core-concepts/agents#cli-backend) or [ACP backend](/core-concepts/agents#acp-backend) cannot spawn Fabro sub-agents. ## Tools diff --git a/docs/public/agents/tools.mdx b/docs/public/agents/tools.mdx index 1c9b6a7c9..5185cfc18 100644 --- a/docs/public/agents/tools.mdx +++ b/docs/public/agents/tools.mdx @@ -6,7 +6,7 @@ description: "Built-in tools for file I/O, shell commands, search, and web acces Every agent in Fabro has access to a set of built-in tools for interacting with the codebase and environment. Tools execute inside the agent's [sandbox](/execution/environments) — whether that's the local machine, a Docker container, or a Daytona VM — so the same tool calls work identically regardless of provider. -The tools described on this page apply to the **API backend** (the default). When using the [CLI backend](/core-concepts/agents#cli-backend), the external CLI tool (`claude`, `codex`, or `gemini`) provides its own tools — Fabro's built-in tools are not used. +The tools described on this page apply to the **API backend** (the default). When using the [CLI backend](/core-concepts/agents#cli-backend) or [ACP backend](/core-concepts/agents#acp-backend), the external agent process provides its own tools — Fabro's built-in tools are not used. ## Core tools diff --git a/docs/public/changelog/2026-02-27.mdx b/docs/public/changelog/2026-02-27.mdx index 67ee8ef40..acc252a59 100644 --- a/docs/public/changelog/2026-02-27.mdx +++ b/docs/public/changelog/2026-02-27.mdx @@ -13,11 +13,11 @@ Two new built-in tools bring real-time information into workflow decisions. `web ## CLI backends -Individual workflow nodes can now delegate work to external AI coding assistants. Set the backend to `claude-code`, `codex`, or `gemini-cli` and the node will use that CLI tool instead of the built-in agent loop. +Individual workflow nodes can delegate work to external AI coding assistants. Set `backend="cli"` and choose a provider; Fabro selects `claude`, `codex`, or `gemini` for the node instead of the built-in API agent loop. Current backend values are `api`, `cli`, and `acp`. ```dot -implement [handler=codergen, cli_backend=codex] -review [handler=codergen, cli_backend=claude-code] +implement [type="agent", backend="cli", provider="openai"] +review [type="agent", backend="cli", provider="anthropic"] ``` This means each stage in a workflow can use a different AI tool — use Codex for implementation and Claude Code for review, for example. diff --git a/docs/public/changelog/2026-05-08.mdx b/docs/public/changelog/2026-05-08.mdx index 613debc40..37e8188d4 100644 --- a/docs/public/changelog/2026-05-08.mdx +++ b/docs/public/changelog/2026-05-08.mdx @@ -1,5 +1,5 @@ --- -title: "Run Events, artifacts, and typed interviews" +title: "Run Events, sandbox lifecycle, and typed interviews" date: "2026-05-08" --- @@ -18,21 +18,26 @@ Run detail sidebars now include dedicated `Run Events` and `Artifacts` pages. `R This gives you a direct path to the two things users usually need after a run: the complete audit trail and the files produced by stages. +## Run-owned sandbox lifecycle + +Run sandboxes now belong to the run lifecycle instead of being detached provider resources that require manual cleanup. Terminal runs stop their sandboxes by default, resumes reconnect to persisted sandboxes, and run deletion either removes the sandbox or returns preserved provider details according to the run's preserve settings. + +The run settings page now shows `stop_on_terminal` alongside sandbox preservation, so you can tell whether cleanup is automatic before launching or deleting a run. + ## Typed interview answers Human-in-the-loop answers now have explicit API shapes for yes/no, single-select, multi-select, and text responses. The web UI and CLI attach flow use those shapes consistently, which removes ambiguity around which field should be present for each question type. `fabro attach` also handles interview edge cases better. Invalid input re-prompts instead of dropping the interaction, and attach no longer blocks when a question was already answered from another client. -## API client coverage for the web app - -The web app now uses the generated TypeScript API client for the auth, workflow, run, artifact, settings, and human-in-the-loop calls that can be generated from OpenAPI. The API reference now documents the browser auth and workflow routes the app relies on, which keeps the UI and public contract aligned as routes evolve. - ## More +- `DELETE /api/v1/runs/{id}` can now return preserved sandbox details when deletion leaves a provider resource alive +- `RunSandboxSettings` now includes `stop_on_terminal` - `RunStage` now includes a canonical `handler` field so clients can choose agent, command, or debug renderers without guessing from names - Run summaries now include stored pull request records with `html_url` +- The web app now uses the generated TypeScript API client for auth, workflow, run, artifact, settings, and human-in-the-loop calls generated from OpenAPI - OpenAPI now documents browser auth config, current user, demo toggle, and development-token login endpoints - OpenAPI now documents workflow list, workflow detail, and workflow runs endpoints - `GET /api/v1/runs/{id}/graph` now documents the optional `direction` query parameter @@ -49,6 +54,8 @@ The web app now uses the generated TypeScript API client for the auth, workflow, - Added a Profile page to the user menu +- Run titles now render inline Markdown in run lists and run detail headers +- Run Files sidebar now shows the whole tree instead of filtering to changed files only - Run cards now show a pull request icon and number only when a stored pull request exists - Run cards now show repository names without repeating the owner prefix - Archived runs now have a Delete action in the web app @@ -56,6 +63,8 @@ The web app now uses the generated TypeScript API client for the auth, workflow, +- Fixed provider token usage normalization so OpenAI, Gemini, and Anthropic totals match provider billing semantics +- Fixed terminal sandbox regressions after run-owned sandbox lifecycle changes - Fixed prompt-only stages missing completed responses in the Thread view - Fixed stage interrupt events missing from the Thread view - Fixed OpenAI reasoning metadata being lost across stateless round trips diff --git a/docs/public/changelog/2026-05-09.mdx b/docs/public/changelog/2026-05-09.mdx new file mode 100644 index 000000000..9e134a541 --- /dev/null +++ b/docs/public/changelog/2026-05-09.mdx @@ -0,0 +1,78 @@ +--- +title: "Sparse inputs, sandbox terminals, and run file diffs" +date: "2026-05-09" +--- + + +**Automatic retros have been removed.** Workflow runs now go directly from execution to finalization and optional pull request creation. The retro crate, retro events, retro docs page, PR retro section, `--no-retro`, `[run.execution].retros`, `features.retros`, and retro projection fields are no longer part of the product surface. + +To migrate: +1. Use Run Events, Stages, Turns, logs, and dumps for post-run analysis. +2. Remove retro-specific CLI flags, config keys, API field reads, and docs links. + + + +**Public local worktree mode has been removed.** Local sandbox runs now execute directly in the resolved working directory, and `--in-place` plus `run.sandbox.local.worktree_mode` are no longer supported. + +To migrate: +1. Run with `--sandbox local` from the checkout you want Fabro to use. +2. Create a separate clone or Git worktree yourself when you want local isolation. +3. Use Docker or Daytona for managed sandbox isolation. + + +## Sparse input overrides + +Workflow inputs can now be overridden one key at a time from the CLI. Repeat `-I` or `--input` on `fabro run`, `fabro create`, and `fabro preflight` to replace specific inputs while preserving unrelated inherited values from settings and workflow config. + +```bash +fabro run .fabro/workflows/check/workflow.toml -I repo_name=fabro-2 --input language=rust +``` + +Input-driven prompt paths, imports, and child workflow paths are rendered with the effective inputs before bundling, so sparse overrides work even when they change which files a workflow references. + +## Sandbox access from run detail + +Run detail now has a dedicated Sandbox tab and terminal route. You can inspect the run sandbox from the web app, open an interactive terminal, copy Docker exec access commands, and see sandbox identity information without switching to a separate CLI session. + +This also gives long-running debug sessions a clearer place to live. Terminal framing, scroll behavior, Daytona proxying, and terminal error states were tightened so the terminal stays usable inside the run UI. + +## Run file diff controls + +The Files Changed view can now compare committed run history and sandbox file scopes from the same toolbar. You can switch diff scope next to the file count, pick commits from the run history, and refresh patch diffs when the selected scope changes. + +This makes the file browser useful both during active sandbox work and after a run has committed checkpoints. + +## More + + +- Run file APIs now expose commit lists and diff scope metadata for committed history and sandbox comparisons +- Run payloads, events, and mutations now support explicit run titles +- Billing projections now include live per-stage token usage while a run is active +- Sandbox details now expose control-plane metadata used by the run Sandbox tab + + + +- Added repeatable `-I, --input ` to `fabro run`, `fabro create`, and `fabro preflight` +- Added Docker host preflight checks for Docker-based deployment and sandbox diagnostics + + + +- Run titles can now be edited inline from the run header +- Run title fields now preserve explicit titles across create, fork, archive, unarchive, and attach flows +- Added a demo-only Start tab for app-shell demos +- Docker Compose deployment defaults now include safer sandbox-facing defaults +- Added Thread DNA timeline strips to debug events and agent stage views +- Added specialized stage renderers for non-agent workflow handlers +- The Billing tab now shows an empty state when no models were used + + + +- Fixed deletion of unreadable runs +- Fixed CLI verification failure handling +- Fixed Daytona terminals to use the toolbox proxy and hide control frames +- Fixed terminal dock spacing, bottom-row clipping, and overlap with the steer bar +- Fixed run file diffs refreshing when the selected scope changes +- Fixed scoped run file diffs to only include tracked files +- Fixed deprecated project directory settings being applied to workflow discovery +- Fixed embedded build Git metadata being stale on branch commits + diff --git a/docs/public/changelog/2026-05-10.mdx b/docs/public/changelog/2026-05-10.mdx new file mode 100644 index 000000000..c8dbe5cbc --- /dev/null +++ b/docs/public/changelog/2026-05-10.mdx @@ -0,0 +1,57 @@ +--- +title: "Sandbox tools, auth sessions, and Live Events" +date: "2026-05-10" +--- + + +**Run API responses now use the canonical `Run` payload.** Run list, board, create, retrieve, and lifecycle endpoints now return the same public run shape. Archive state moved out of `status.kind = "archived"` and into run lifecycle metadata, sandbox fields distinguish planned settings from runtime state, and pull request records are separate from live pull request details. + +To migrate: +1. Regenerate API clients from the current OpenAPI spec. +2. Replace `RunSummary`, `RunListItem`, and `RunStatusResponse` assumptions with the canonical `Run` shape. +3. Read archive state from lifecycle metadata instead of treating `archived` as a terminal status. + + +## Sandbox tools in one place + +The run Sandbox tab now includes a read-only filesystem browser, Daytona VNC previews, discovered sandbox services, and terminal access from the same page. Empty files render cleanly, large file previews are virtualized, and directory sentinel files are hidden from the browser. + +Daytona sandboxes can start a signed noVNC preview from the web UI, and discovered listening TCP services are listed with their ports, bind addresses, process summaries, and preview support. + +## Auth sessions and Live Events + +Profile now includes a Sessions page for the current browser session and active CLI session chains. CLI sessions can be revoked from the same surface, while browser sessions are listed as non-revocable in this API version. + +Settings now includes a Live Events page for watching server events in the web app. The page reuses the event debugger and adds the filtering and navigation needed for operational inspection. + +## Automations navigation + +The web app now uses Automations as the product label for workflow definitions and runs. Routes, nav labels, and run pages were updated together so the main app shell reads as Automations, Runs, Settings, and Profile instead of mixing workflow terminology into the navigation. + +## More + + +- New `GET /api/v1/auth/sessions` endpoint lists browser and CLI auth sessions for the authenticated user +- New `DELETE /api/v1/auth/sessions/{id}` endpoint revokes active CLI session chains +- New `GET /api/v1/runs/{id}/sandbox/services` endpoint lists listening TCP services inside a run sandbox +- New `POST /api/v1/runs/{id}/sandbox/vnc` endpoint creates signed noVNC preview URLs for Daytona sandboxes +- Run list, board, create, retrieve, and lifecycle endpoints now return canonical `Run` objects + + + +- Added Profile sub-navigation with Overview and Sessions pages +- Added Settings sub-navigation with Overview and Live Events pages +- Added a full-screen terminal route and an "Open in new tab" action for embedded terminals +- Added read-only filesystem and Daytona VNC modes to the run Sandbox tab +- Added a Services tab to the run Sandbox page +- Virtualized sandbox file previews and improved empty-file handling +- Renamed the Workflows tab to Automations + + + +- Fixed sandbox service discovery so listening services are detected more reliably +- Fixed sandbox VNC previews to open the noVNC viewer page +- Fixed full-screen terminal toast notifications by wrapping the route in the Toast provider +- Fixed filesystem directory sentinels showing in the sandbox file browser +- Fixed projection cache hydration before appending later events + diff --git a/docs/public/changelog/2026-05-11.mdx b/docs/public/changelog/2026-05-11.mdx new file mode 100644 index 000000000..cdccc384e --- /dev/null +++ b/docs/public/changelog/2026-05-11.mdx @@ -0,0 +1,47 @@ +--- +title: "Fabro MCP server" +date: "2026-05-11" +--- + +## Fabro MCP server + +Fabro now ships a stdio-based Model Context Protocol server, so MCP clients can manage workflow runs through the authenticated `fabro` CLI. It reuses normal CLI server targeting, OAuth refresh, dev-token and local-server handling, proxy behavior, and storage configuration instead of requiring a separate MCP authentication flow. + +```bash +fabro mcp init claude +``` + +You can also print client configuration JSON or start the server directly: + +```bash +fabro mcp config +fabro mcp start +``` + +## Run management from MCP clients + +The MCP server exposes structured tools for creating runs, searching runs, reading run events, interacting with pending human questions, and waiting for runs to finish. MCP-created runs use the same manifest construction and override semantics as CLI-created runs, so workflow paths, goals, inputs, labels, model settings, sandbox settings, and dry-run options behave consistently. + +This lets agent tools orchestrate Fabro runs without scraping CLI output or hand-rolling HTTP clients. + +## More + + +- Added `fabro mcp start` to launch the MCP server over stdio +- Added `fabro mcp config` to print MCP client configuration JSON +- Added `fabro mcp init ` for Claude, Cursor, and Windsurf client setup + + + +- Added a bundled `daytona-medium` workflow for verifying the Daytona Medium sandbox starts with standard tooling + + + +- MCP run tools can create multiple runs, apply scalar input overrides, attach labels, start or stage runs, and return structured run summaries +- MCP event tools support category, event type, text, timestamp, direction, and pagination filters +- MCP interact tools support cancelling, archiving, unarchiving, inspecting questions, answering questions, and sending run messages + + + +- Fixed deleting terminal runs so successful completions are not followed by cancelled failure events + diff --git a/docs/public/core-concepts/agents.mdx b/docs/public/core-concepts/agents.mdx index e3d4abf7e..8a4331d4f 100644 --- a/docs/public/core-concepts/agents.mdx +++ b/docs/public/core-concepts/agents.mdx @@ -18,7 +18,7 @@ This loop continues until the model stops calling tools, indicating it considers ## Backends -Every agent node uses a **backend** that determines how Fabro interacts with the LLM. There are two options: +Every agent and prompt node uses a **backend** that determines how Fabro interacts with the LLM. There are three options: ### API backend (default) @@ -31,7 +31,7 @@ Fabro manages the agent loop directly — it calls the LLM provider's API, execu ### CLI backend -Fabro delegates execution to an external coding assistant CLI. The CLI tool manages its own tool loop internally — Fabro sends the prompt, waits for the CLI to finish, and tracks file changes via `git diff` before and after execution. +Fabro delegates execution to a legacy external coding assistant CLI. The CLI tool manages its own tool loop internally — Fabro sends the prompt, waits for the CLI to finish, and tracks file changes via `git diff` before and after execution. The CLI is selected automatically based on the node's provider: @@ -41,6 +41,8 @@ The CLI is selected automatically based on the node's provider: | OpenAI | `codex` | | Gemini | `gemini` | +Fabro does not install these CLIs at runtime. Install the selected CLI in the sandbox image or run setup steps before the workflow reaches a `backend="cli"` node. + Set the CLI backend on a node with `backend="cli"` or via a [model stylesheet](/workflows/stylesheets): ```dot @@ -52,15 +54,29 @@ implement [label="Implement", backend="cli"] * { backend: cli; } ``` +### ACP backend + +Fabro can also run Agent Client Protocol (ACP) stdio agents with `backend="acp"`. ACP agents run inside the active Fabro sandbox, so local and Docker runs keep the same workspace isolation, secret forwarding, cancellation, and file-change tracking behavior as other agent stages. + +Set ACP on a node with `backend="acp"` and an explicit `acp_command`: + +```dot +implement [label="Implement", backend="acp", acp_command="python3 tools/fake_acp_agent.py"] +``` + +Fabro does not install ACP agents, Node.js, npm, or `npx` at runtime. The command must already be available in the sandbox image, repository, or setup steps. You can use `npx ...@latest` as an explicit `acp_command` if that is the behavior you want, but Fabro will treat it like any other user-supplied command. + +ACP v1 does not have a portable model-selection request. Fabro records the selected provider and model in events and run projections, but model-specific ACP behavior must be encoded in the chosen command for now. ACP is supported with local and Docker sandboxes; Daytona does not expose bidirectional stdio yet, so ACP nodes fail there with an explicit unsupported-provider error. + ### Comparison -| Capability | API backend | CLI backend | -|---|---|---| -| Tools | Fabro built-in tools + MCP | CLI's own tool set | -| Session caching | Supported (`fidelity` + `thread_id`) | Not supported | -| Sub-agents | Supported | Not supported | -| Provider failover | Supported | Not supported | -| File tracking | Tool call events | `git diff` before/after | +| Capability | API backend | CLI backend | ACP backend | +|---|---|---|---| +| Tools | Fabro built-in tools + MCP | CLI's own tool set | ACP agent's own tool set | +| Session caching | Supported (`fidelity` + `thread_id`) | Not supported | Agent-dependent | +| Sub-agents | Supported | Not supported | Not supported through Fabro tools | +| Provider failover | Supported | Not supported | Not supported | +| File tracking | Tool call events | `git diff` before/after | `git diff` before/after | ### When to use the CLI backend @@ -68,6 +84,12 @@ implement [label="Implement", backend="cli"] - **CLI-only models** — use models that are only available through a CLI tool, not via API - **Existing workflows** — integrate a CLI tool you already depend on without rewriting its configuration +### When to use the ACP backend + +- **Protocol adapters** — run ACP-compatible coding agents through a stable stdio protocol +- **Sandbox parity** — keep agent process execution inside Fabro's local or Docker sandbox +- **Custom agents** — use `acp_command` for a checked-in or preinstalled ACP adapter + ## Tools Agents have access to a set of built-in tools for interacting with the codebase and environment: diff --git a/docs/public/docs.json b/docs/public/docs.json index 3055c1b8e..b7039ca11 100644 --- a/docs/public/docs.json +++ b/docs/public/docs.json @@ -250,6 +250,9 @@ "group": "May 2026", "icon": "clock-rotate-left", "pages": [ + "changelog/2026-05-11", + "changelog/2026-05-10", + "changelog/2026-05-09", "changelog/2026-05-08", "changelog/2026-05-07", "changelog/2026-05-06", diff --git a/docs/public/reference/cli.mdx b/docs/public/reference/cli.mdx index 8d9f1f939..7d5bfc2a7 100644 --- a/docs/public/reference/cli.mdx +++ b/docs/public/reference/cli.mdx @@ -79,6 +79,7 @@ fabro [OPTIONS] [COMMAND] | `fabro inspect` | Show detailed information about a workflow run | | `fabro install` | Set up the Fabro environment (LLMs, certs, GitHub) | | `fabro logs` | View the raw worker tracing log of a workflow run | +| `fabro mcp` | Model Context Protocol server | | `fabro model` | List and test LLM models | | `fabro pr` | Pull request operations | | `fabro preflight` | Validate run configuration without executing | @@ -510,6 +511,73 @@ fabro logs [OPTIONS] | `--server ` | Fabro server target: http(s) URL or absolute Unix socket path | | `-n, --tail ` | Lines from end (default: all) | +### `fabro mcp` + +Model Context Protocol server + +```bash +fabro mcp [OPTIONS] +``` + +#### Subcommands + +| Command | Description | +| --- | --- | +| `fabro mcp config` | Print MCP client configuration JSON | +| `fabro mcp init` | Configure an MCP client to launch Fabro | +| `fabro mcp start` | Start the Fabro MCP server over stdio | + +#### `fabro mcp config` + +Print MCP client configuration JSON + +```bash +fabro mcp config [OPTIONS] +``` + +#### Options + +| Option | Description | +| --- | --- | +| `--server ` | Fabro server target: http(s) URL or absolute Unix socket path | +| `--storage-dir ` | Local storage directory (default: ~/.fabro/storage) | + +#### `fabro mcp init` + +Configure an MCP client to launch Fabro + +```bash +fabro mcp init [OPTIONS] +``` + +#### Arguments + +| Name | Description | +| --- | --- | +| `AGENT` | Values: `claude`, `cursor`, `windsurf` | + +#### Options + +| Option | Description | +| --- | --- | +| `--server ` | Fabro server target: http(s) URL or absolute Unix socket path | +| `--storage-dir ` | Local storage directory (default: ~/.fabro/storage) | + +#### `fabro mcp start` + +Start the Fabro MCP server over stdio + +```bash +fabro mcp start [OPTIONS] +``` + +#### Options + +| Option | Description | +| --- | --- | +| `--server ` | Fabro server target: http(s) URL or absolute Unix socket path | +| `--storage-dir ` | Local storage directory (default: ~/.fabro/storage) | + ### `fabro model` List and test LLM models diff --git a/docs/public/reference/dot-language.mdx b/docs/public/reference/dot-language.mdx index 12436ec70..3ba7a7259 100644 --- a/docs/public/reference/dot-language.mdx +++ b/docs/public/reference/dot-language.mdx @@ -206,7 +206,8 @@ Start nodes can also be identified by ID (`start` or `Start`). Exit nodes can be | `model` | String | Explicit model ID (overrides stylesheet) | | `provider` | String | Explicit provider name (overrides stylesheet). Auto-inferred from the model catalog when omitted. | | `project_memory` | Boolean | When `true` (default), prompt nodes discover and include project docs (`AGENTS.md`, `CLAUDE.md`, etc.) as a system prompt. Set to `false` to disable. | -| `backend` | String | Agent execution backend. `api` (default): Fabro calls the LLM API directly and runs its own tool loop. `cli`: Fabro delegates to an external CLI tool (`claude`, `codex`, or `gemini` based on provider). See [Agents — Backends](/core-concepts/agents#backends). | +| `backend` | String | Agent execution backend: `api` (default), `cli`, or `acp`. `api` runs Fabro's tool loop through provider APIs; `cli` delegates to the legacy provider CLI; `acp` runs an Agent Client Protocol stdio agent inside the active sandbox. See [Agents — Backends](/core-concepts/agents#backends). | +| `acp_command` | String | Required for nodes with `backend="acp"`. The value must be a stdio ACP command available in the sandbox. Fabro records model selection but does not send it through stable ACP v1. | ### Command nodes diff --git a/docs/public/tutorials/multi-model.mdx b/docs/public/tutorials/multi-model.mdx index 9f444ddcb..16567d597 100644 --- a/docs/public/tutorials/multi-model.mdx +++ b/docs/public/tutorials/multi-model.mdx @@ -88,7 +88,7 @@ Stylesheets support four properties: | `model` | Model ID or alias (e.g. `claude-sonnet-4-5`, `opus`, `gemini-pro`) | | `provider` | Provider name (optional — auto-inferred from the model catalog when omitted) | | `reasoning_effort` | `low`, `medium`, or `high` | -| `backend` | `api` (default) or `cli` | +| `backend` | `api` (default), `cli`, or `acp` | ## Why route models? diff --git a/docs/public/workflows/stylesheets.mdx b/docs/public/workflows/stylesheets.mdx index 6f171f92e..c1035b430 100644 --- a/docs/public/workflows/stylesheets.mdx +++ b/docs/public/workflows/stylesheets.mdx @@ -68,7 +68,7 @@ Stylesheets support four properties: | `provider` | Provider name (optional — auto-inferred from the model catalog when omitted) | `anthropic`, `openai`, `gemini` | | `reasoning_effort` | Reasoning effort level | `low`, `medium`, `high` | | `speed` | Output speed mode. `fast` enables Anthropic's fast mode for up to 2.5x faster output at higher cost. | `fast` | -| `backend` | Agent execution backend — `api` (default) runs Fabro's own tool loop, `cli` delegates to an external CLI tool. See [Backends](/core-concepts/agents#backends). | `cli`, `api` | +| `backend` | Agent execution backend — `api` (default) runs Fabro's own tool loop, `cli` delegates to a legacy external CLI tool, and `acp` runs an Agent Client Protocol stdio agent in the active sandbox. See [Backends](/core-concepts/agents#backends). | `api`, `cli`, `acp` | See [Models](/core-concepts/models) for the full list of model IDs and aliases. diff --git a/lib/crates/fabro-acp/Cargo.toml b/lib/crates/fabro-acp/Cargo.toml new file mode 100644 index 000000000..c373c7944 --- /dev/null +++ b/lib/crates/fabro-acp/Cargo.toml @@ -0,0 +1,37 @@ +[package] +name = "fabro-acp" +edition.workspace = true +version.workspace = true +publish = false +license.workspace = true +description = "Agent Client Protocol backend support for Fabro" + +[features] +test-support = [] + +[lib] +doctest = false + +[lints] +workspace = true + +[dependencies] +agent-client-protocol.workspace = true +agent-client-protocol-tokio.workspace = true +fabro-model = { path = "../fabro-model" } +fabro-sandbox = { path = "../fabro-sandbox" } +fabro-types = { path = "../fabro-types" } +fabro-util = { path = "../fabro-util" } +bytes.workspace = true +serde.workspace = true +serde_json.workspace = true +thiserror.workspace = true +tokio.workspace = true +tokio-util = { workspace = true, features = ["compat", "io"] } +futures.workspace = true +uuid.workspace = true +tracing.workspace = true + +[dev-dependencies] +fabro-sandbox = { path = "../fabro-sandbox", features = ["test-support"] } +tempfile = "3" diff --git a/lib/crates/fabro-acp/src/command.rs b/lib/crates/fabro-acp/src/command.rs new file mode 100644 index 000000000..c27ecd35f --- /dev/null +++ b/lib/crates/fabro-acp/src/command.rs @@ -0,0 +1,190 @@ +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::str::FromStr; + +use agent_client_protocol::schema::McpServer; +use agent_client_protocol_tokio::AcpAgent; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AcpCommand { + display: String, + program: PathBuf, + args: Vec, + env: HashMap, +} + +impl AcpCommand { + #[must_use] + pub fn program(&self) -> &Path { + &self.program + } + + #[must_use] + pub fn args(&self) -> &[String] { + &self.args + } + + #[must_use] + pub fn env(&self) -> &HashMap { + &self.env + } + + #[must_use] + pub fn display(&self) -> &str { + &self.display + } + + #[must_use] + pub fn to_shell_command(&self) -> String { + render_command(&self.program, &self.args) + } +} + +impl std::fmt::Display for AcpCommand { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.display) + } +} + +#[derive(Debug, thiserror::Error)] +pub enum AcpCommandError { + #[error("acp_command must not be empty")] + EmptyOverride, + #[error( + "acp_command is required for backend=\"acp\" because Fabro does not install ACP agents" + )] + MissingOverride, + #[error("only stdio ACP commands are supported")] + UnsupportedTransport, + #[error("failed to parse acp_command")] + Parse(#[source] agent_client_protocol::Error), +} + +impl From for AcpCommandError { + fn from(error: agent_client_protocol::Error) -> Self { + Self::Parse(error) + } +} + +pub fn resolve_acp_command(override_command: Option<&str>) -> Result { + if let Some(raw) = override_command { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return Err(AcpCommandError::EmptyOverride); + } + return parse_acp_command(trimmed); + } + + Err(AcpCommandError::MissingOverride) +} + +fn parse_acp_command(raw: &str) -> Result { + reject_non_stdio_json_transport(raw)?; + + let agent = AcpAgent::from_str(raw)?; + let McpServer::Stdio(stdio) = agent.into_server() else { + return Err(AcpCommandError::UnsupportedTransport); + }; + + let program = stdio.command; + let args = stdio.args; + let display = render_command(&program, &args); + + Ok(AcpCommand { + display, + program, + args, + env: stdio + .env + .into_iter() + .map(|env| (env.name, env.value)) + .collect(), + }) +} + +fn render_command(program: &Path, args: &[String]) -> String { + std::iter::once(program.to_string_lossy().into_owned()) + .chain(args.iter().cloned()) + .map(|part| fabro_sandbox::shell_quote(&part)) + .collect::>() + .join(" ") +} + +fn reject_non_stdio_json_transport(raw: &str) -> Result<(), AcpCommandError> { + let trimmed = raw.trim_start(); + if !trimmed.starts_with('{') { + return Ok(()); + } + + let Ok(value) = serde_json::from_str::(trimmed) else { + return Ok(()); + }; + + match value.get("type").and_then(serde_json::Value::as_str) { + Some("stdio") | None => Ok(()), + Some(_) => Err(AcpCommandError::UnsupportedTransport), + } +} + +#[cfg(test)] +mod tests { + use std::path::Path; + + use super::*; + + #[test] + fn missing_acp_command_is_rejected() { + let err = resolve_acp_command(None).unwrap_err(); + assert!( + err.to_string() + .contains("acp_command is required for backend=\"acp\"") + ); + } + + #[test] + fn explicit_acp_command_overrides_provider_default() { + let command = resolve_acp_command(Some("python fake_agent.py")).unwrap(); + assert_eq!(command.to_string(), "python fake_agent.py"); + assert_eq!(command.program(), Path::new("python")); + assert_eq!(command.args(), &["fake_agent.py".to_string()]); + } + + #[test] + fn blank_acp_command_is_rejected() { + let err = resolve_acp_command(Some(" ")).unwrap_err(); + assert!(err.to_string().contains("acp_command must not be empty")); + } + + #[test] + fn json_stdio_acp_command_is_supported() { + let raw = r#"{"type":"stdio","name":"fake","command":"python","args":["fake agent.py"],"env":[{"name":"MODE","value":"test"}]}"#; + let command = resolve_acp_command(Some(raw)).unwrap(); + assert_eq!(command.program(), Path::new("python")); + assert_eq!(command.args(), &["fake agent.py".to_string()]); + assert_eq!(command.env().get("MODE").map(String::as_str), Some("test")); + } + + #[test] + fn json_stdio_acp_command_display_omits_env_contents() { + let raw = r#"{"type":"stdio","name":"fake","command":"agent","args":["--flag","two words"],"env":[{"name":"OPENAI_API_KEY","value":"secret-key"}]}"#; + let command = resolve_acp_command(Some(raw)).unwrap(); + + assert_eq!( + command.env().get("OPENAI_API_KEY").map(String::as_str), + Some("secret-key") + ); + assert_eq!(command.to_string(), "agent --flag 'two words'"); + assert!(!command.to_string().contains("secret-key")); + assert!(!command.to_string().contains("OPENAI_API_KEY")); + } + + #[test] + fn non_stdio_acp_command_is_rejected() { + let raw = r#"{"type":"http","name":"remote","url":"https://example.test/acp"}"#; + let err = resolve_acp_command(Some(raw)).unwrap_err(); + assert!( + err.to_string() + .contains("only stdio ACP commands are supported") + ); + } +} diff --git a/lib/crates/fabro-acp/src/error.rs b/lib/crates/fabro-acp/src/error.rs new file mode 100644 index 000000000..37ae511e5 --- /dev/null +++ b/lib/crates/fabro-acp/src/error.rs @@ -0,0 +1,31 @@ +use crate::command::AcpCommandError; + +#[derive(Debug, thiserror::Error)] +pub enum AcpError { + #[error(transparent)] + Command(#[from] AcpCommandError), + + #[error(transparent)] + Sandbox(#[from] fabro_sandbox::Error), + + #[error("ACP protocol error")] + Protocol(#[source] agent_client_protocol::Error), + + #[error("ACP turn was cancelled")] + Cancelled, + + #[error("ACP turn timed out")] + TimedOut { stderr: String }, + + #[error("ACP prompt stopped with {stop_reason}: {text}")] + StopReason { + stop_reason: String, + text: String, + }, +} + +impl From for AcpError { + fn from(error: agent_client_protocol::Error) -> Self { + Self::Protocol(error) + } +} diff --git a/lib/crates/fabro-acp/src/lib.rs b/lib/crates/fabro-acp/src/lib.rs new file mode 100644 index 000000000..41f4d2fbe --- /dev/null +++ b/lib/crates/fabro-acp/src/lib.rs @@ -0,0 +1,12 @@ +pub mod command; +pub mod error; +pub mod session; + +#[cfg(any(test, feature = "test-support"))] +pub mod test_support; + +mod transport; + +pub use command::{AcpCommand, AcpCommandError, resolve_acp_command}; +pub use error::AcpError; +pub use session::{AcpRunRequest, AcpRunResult, render_stop_reason, run_acp_turn}; diff --git a/lib/crates/fabro-acp/src/session.rs b/lib/crates/fabro-acp/src/session.rs new file mode 100644 index 000000000..bdff151d3 --- /dev/null +++ b/lib/crates/fabro-acp/src/session.rs @@ -0,0 +1,268 @@ +use std::collections::HashMap; +use std::sync::Arc; +use std::time::Duration; + +use agent_client_protocol::schema::{ + CancelNotification, ContentBlock, ContentChunk, InitializeRequest, PermissionOptionKind, + ProtocolVersion, RequestPermissionOutcome, RequestPermissionRequest, RequestPermissionResponse, + SelectedPermissionOutcome, SessionNotification, SessionUpdate, StopReason, +}; +use agent_client_protocol::util::MatchDispatch; +use agent_client_protocol::{ActiveSession, Agent, Client, Error as ProtocolError, SessionMessage}; +use fabro_sandbox::Sandbox; +use fabro_util::time::elapsed_ms; +use tokio::time::{sleep, timeout}; +use tokio_util::sync::CancellationToken; + +use crate::command::AcpCommand; +use crate::error::AcpError; +use crate::transport::{SandboxAcpTransport, TransportState}; + +pub struct AcpRunRequest { + pub command: AcpCommand, + pub prompt: String, + pub cwd: String, + pub timeout_ms: Option, + pub env: HashMap, + pub sandbox: Arc, + pub cancel_token: CancellationToken, + pub on_activity: Option>, +} + +#[derive(Debug)] +pub struct AcpRunResult { + pub text: String, + pub stop_reason: StopReason, + pub stderr: String, + pub duration_ms: u64, +} + +pub async fn run_acp_turn(request: AcpRunRequest) -> Result { + let AcpRunRequest { + command, + prompt, + cwd, + timeout_ms, + env, + sandbox, + cancel_token, + on_activity, + } = request; + let start = std::time::Instant::now(); + let state = TransportState::new(); + let read_cancel_token = cancel_token.clone(); + let run_cancel_token = cancel_token.clone(); + let permission_cancel_token = cancel_token.clone(); + let state_for_run = state.clone(); + let transport = SandboxAcpTransport::new(command, cwd.clone(), env, sandbox, state.clone()); + + let run = Client + .builder() + .name("fabro") + .on_receive_request( + async move |request: RequestPermissionRequest, responder, _connection| { + let outcome = if permission_cancel_token.is_cancelled() { + RequestPermissionOutcome::Cancelled + } else { + select_permission_outcome(&request) + }; + responder.respond(RequestPermissionResponse::new(outcome)) + }, + agent_client_protocol::on_receive_request!(), + ) + .connect_with(transport, async move |cx| { + cx.send_request(InitializeRequest::new(ProtocolVersion::V1)) + .block_task() + .await?; + + cx.build_session(&cwd) + .block_task() + .run_until(async |mut session| { + session.send_prompt(prompt)?; + read_turn( + &mut session, + &read_cancel_token, + on_activity.as_ref(), + &state_for_run, + ) + .await + }) + .await + }); + + let cancel_deadline_token = cancel_token.clone(); + let run_outcome = async { + match timeout_ms { + Some(timeout_ms) => { + if let Ok(result) = timeout(Duration::from_millis(timeout_ms), run).await { + Ok(result) + } else { + state.terminate().await?; + if run_cancel_token.is_cancelled() { + return Err(AcpError::Cancelled); + } + Err(AcpError::TimedOut { + stderr: state.stderr_tail().await, + }) + } + } + None => Ok(run.await), + } + }; + let outcome = tokio::select! { + result = run_outcome => result?, + () = async { + cancel_deadline_token.cancelled().await; + sleep(Duration::from_millis(500)).await; + } => { + state.terminate().await?; + return Err(AcpError::Cancelled); + } + }; + let (text, stop_reason) = match outcome { + Ok(result) => result, + Err(_) if run_cancel_token.is_cancelled() => { + state.terminate().await?; + return Err(AcpError::Cancelled); + } + Err(error) => { + state.terminate().await?; + if let Some(startup_error) = state.take_startup_error().await { + return Err(AcpError::Sandbox(startup_error)); + } + return Err(map_protocol_error(error)); + } + }; + + match stop_reason { + StopReason::EndTurn | StopReason::Refusal => {} + StopReason::Cancelled => { + state.terminate().await?; + return Err(AcpError::Cancelled); + } + _ => { + state.terminate().await?; + return Err(AcpError::StopReason { + stop_reason: render_stop_reason(&stop_reason), + text, + }); + } + } + + state.terminate().await?; + let stderr = state.stderr_tail().await; + Ok(AcpRunResult { + text, + stop_reason, + stderr, + duration_ms: elapsed_ms(start), + }) +} + +fn map_protocol_error(error: ProtocolError) -> AcpError { + AcpError::Protocol(error) +} + +fn select_permission_outcome(request: &RequestPermissionRequest) -> RequestPermissionOutcome { + let selected = request + .options + .iter() + .find(|option| option.kind == PermissionOptionKind::AllowAlways) + .or_else(|| { + request + .options + .iter() + .find(|option| option.kind == PermissionOptionKind::AllowOnce) + }) + .or_else(|| { + request.options.iter().find(|option| { + !matches!( + option.kind, + PermissionOptionKind::RejectOnce | PermissionOptionKind::RejectAlways + ) + }) + }); + + selected.map_or(RequestPermissionOutcome::Cancelled, |option| { + RequestPermissionOutcome::Selected(SelectedPermissionOutcome::new(option.option_id.clone())) + }) +} + +async fn read_turn( + session: &mut ActiveSession<'_, Agent>, + cancel_token: &CancellationToken, + on_activity: Option<&Arc>, + state: &TransportState, +) -> Result<(String, StopReason), ProtocolError> { + let mut text = String::new(); + let mut cancel_sent = false; + + loop { + tokio::select! { + update = session.read_update() => { + if let Some(on_activity) = on_activity { + on_activity(); + } + match update? { + SessionMessage::SessionMessage(dispatch) => { + MatchDispatch::new(dispatch) + .if_notification(async |notification: SessionNotification| { + if let SessionUpdate::AgentMessageChunk(ContentChunk { + content: ContentBlock::Text(text_chunk), + .. + }) = notification.update { + text.push_str(&text_chunk.text); + } + Ok(()) + }) + .await + .otherwise_ignore()?; + } + SessionMessage::StopReason(stop_reason) => { + return Ok((text, stop_reason)); + } + _ => {} + } + } + () = cancel_token.cancelled(), if !cancel_sent => { + cancel_sent = true; + session.connection().send_notification_to( + Agent, + CancelNotification::new(session.session_id().clone()), + )?; + } + () = sleep(Duration::from_millis(500)), if cancel_sent => { + state.terminate().await.map_err(ProtocolError::into_internal_error)?; + return Ok((text, StopReason::Cancelled)); + } + } + } +} + +#[must_use] +pub fn render_stop_reason(stop_reason: &StopReason) -> String { + serde_json::to_value(stop_reason) + .ok() + .and_then(|value| value.as_str().map(str::to_string)) + .unwrap_or_else(|| format!("{stop_reason:?}")) +} + +#[cfg(test)] +mod tests { + use agent_client_protocol::schema::SessionNotification; + + #[test] + fn codex_usage_update_session_notification_deserializes() { + let notification = serde_json::json!({ + "sessionId": "session-1", + "update": { + "sessionUpdate": "usage_update", + "used": 26128, + "size": 258_400 + } + }); + + serde_json::from_value::(notification) + .expect("Codex ACP usage_update notifications should be ignored, not fatal"); + } +} diff --git a/lib/crates/fabro-acp/src/test_support.rs b/lib/crates/fabro-acp/src/test_support.rs new file mode 100644 index 000000000..fc697731b --- /dev/null +++ b/lib/crates/fabro-acp/src/test_support.rs @@ -0,0 +1,150 @@ +use agent_client_protocol::schema::{ + ContentBlock, ContentChunk, SessionNotification, SessionUpdate, +}; +use serde_json::json; + +pub const SESSION_ID: &str = "sess-1"; + +pub fn agent_message_chunk(session_id: &str, text: &str) -> SessionNotification { + SessionNotification::new( + session_id.to_string(), + SessionUpdate::AgentMessageChunk(ContentChunk::new(ContentBlock::from(text.to_string()))), + ) +} + +pub fn agent_message_chunk_json(session_id: &str, text: &str) -> serde_json::Value { + json!({ + "jsonrpc": "2.0", + "method": "session/update", + "params": agent_message_chunk(session_id, text), + }) +} + +pub fn fake_acp_agent_script() -> &'static str { + r#" +import json +import os +import signal +import sys +import time + +methods = [] +session_id = "sess-1" + +if os.environ.get("ACP_PID_RECORD"): + with open(os.environ["ACP_PID_RECORD"], "w", encoding="utf-8") as record: + record.write(str(os.getpid())) + +def handle_sigterm(signum, frame): + if os.environ.get("ACP_LINGER_TERMINATED"): + with open(os.environ["ACP_LINGER_TERMINATED"], "w", encoding="utf-8") as record: + record.write("terminated\n") + sys.exit(0) + +signal.signal(signal.SIGTERM, handle_sigterm) + +def send(message): + print(json.dumps(message), flush=True) + +def respond(message, result): + send({"jsonrpc": "2.0", "id": message["id"], "result": result}) + +def record_methods(): + if os.environ.get("ACP_RECORD"): + with open(os.environ["ACP_RECORD"], "w", encoding="utf-8") as record: + record.write("\n".join(methods) + "\n") + +for line in sys.stdin: + message = json.loads(line) + method = message.get("method") + methods.append(method) + + if method == "initialize": + if os.environ.get("ACP_MODE") == "slow_initialize": + time.sleep(60) + respond(message, {"protocolVersion": 1, "agentCapabilities": {}}) + elif method == "session/new": + if os.environ.get("ACP_SESSION_NEW_PARAMS"): + with open(os.environ["ACP_SESSION_NEW_PARAMS"], "w", encoding="utf-8") as record: + record.write(json.dumps(message.get("params", {}), separators=(",", ":"))) + respond(message, {"sessionId": session_id}) + elif method == "session/prompt": + if os.environ.get("ACP_PROMPT_RECORD"): + with open(os.environ["ACP_PROMPT_RECORD"], "w", encoding="utf-8") as record: + record.write(json.dumps(message.get("params", {}))) + mode = os.environ.get("ACP_MODE", "normal") + if mode == "timeout": + time.sleep(60) + if mode == "malformed": + print("malformed json", file=sys.stderr, flush=True) + print("{not-json", flush=True) + break + if mode == "early_exit": + print("early boom", file=sys.stderr, flush=True) + sys.exit(2) + if mode == "write_file": + with open("hello.txt", "w", encoding="utf-8") as file: + file.write("hello from sandbox\n") + if mode == "cancel": + send({ + "jsonrpc": "2.0", + "method": "session/update", + "params": { + "sessionId": session_id, + "update": { + "sessionUpdate": "agent_message_chunk", + "content": {"type": "text", "text": "waiting for cancellation"} + } + } + }) + for cancel_line in sys.stdin: + cancel_message = json.loads(cancel_line) + if cancel_message.get("method") == "session/cancel": + with open(os.environ["ACP_CANCEL_RECORD"], "w", encoding="utf-8") as record: + record.write("session/cancel\n") + respond(message, {"stopReason": "cancelled"}) + sys.exit(0) + if mode == "permission": + send({ + "jsonrpc": "2.0", + "id": "permission-1", + "method": "session/request_permission", + "params": { + "sessionId": session_id, + "toolCall": {"toolCallId": "tool-1"}, + "options": [ + {"optionId": "reject", "name": "Reject", "kind": "reject_once"}, + {"optionId": "once", "name": "Allow once", "kind": "allow_once"}, + {"optionId": "always", "name": "Allow always", "kind": "allow_always"} + ] + } + }) + permission_response = json.loads(sys.stdin.readline()) + with open(os.environ["ACP_PERMISSION"], "w", encoding="utf-8") as permission: + permission.write(json.dumps(permission_response.get("result", {}), separators=(",", ":"))) + for text in ["hello ", "from acp"]: + send({ + "jsonrpc": "2.0", + "method": "session/update", + "params": { + "sessionId": session_id, + "update": { + "sessionUpdate": "agent_message_chunk", + "content": {"type": "text", "text": text} + } + } + }) + record_methods() + respond(message, {"stopReason": os.environ.get("ACP_STOP_REASON", "end_turn")}) + if mode == "linger_after_response": + while True: + time.sleep(1) + break + else: + send({ + "jsonrpc": "2.0", + "id": message.get("id"), + "error": {"code": -32601, "message": "method not found"} + }) +"# +} diff --git a/lib/crates/fabro-acp/src/transport.rs b/lib/crates/fabro-acp/src/transport.rs new file mode 100644 index 000000000..e3915c93e --- /dev/null +++ b/lib/crates/fabro-acp/src/transport.rs @@ -0,0 +1,156 @@ +use std::collections::HashMap; +use std::io::Result as IoResult; +use std::pin::Pin; +use std::sync::Arc; +use std::time::Duration; + +use agent_client_protocol::util::internal_error; +use agent_client_protocol::{ + Agent, Client, ConnectTo, Error as ProtocolError, Lines, Result as AcpProtocolResult, +}; +use fabro_sandbox::{ + Error as SandboxError, Result as SandboxResult, Sandbox, StderrCollector, StdioProcessHandle, +}; +use futures::io::BufReader; +use futures::sink::unfold; +use futures::{AsyncBufReadExt, AsyncWriteExt, Stream}; +use tokio::sync::Mutex as TokioMutex; +use tokio::time::timeout; +use tokio_util::compat::{TokioAsyncReadCompatExt, TokioAsyncWriteCompatExt}; + +use crate::command::AcpCommand; + +#[derive(Clone)] +pub(crate) struct TransportState { + handle: Arc>>, + stderr: Arc>>, + startup_error: Arc>>, +} + +impl TransportState { + pub(crate) fn new() -> Self { + Self { + handle: Arc::new(TokioMutex::new(None)), + stderr: Arc::new(TokioMutex::new(None)), + startup_error: Arc::new(TokioMutex::new(None)), + } + } + + async fn set_process(&self, handle: StdioProcessHandle, stderr: StderrCollector) { + *self.handle.lock().await = Some(handle); + *self.stderr.lock().await = Some(stderr); + } + + async fn set_startup_error(&self, error: SandboxError) { + *self.startup_error.lock().await = Some(error); + } + + pub(crate) async fn take_startup_error(&self) -> Option { + self.startup_error.lock().await.take() + } + + pub(crate) async fn terminate(&self) -> SandboxResult<()> { + if let Some(handle) = self.handle.lock().await.as_ref().cloned() { + handle.terminate().await?; + } + Ok(()) + } + + pub(crate) async fn stderr_tail(&self) -> String { + if let Some(stderr) = self.stderr.lock().await.as_ref().cloned() { + return stderr.tail_string().await; + } + String::new() + } +} + +pub(crate) struct SandboxAcpTransport { + command: AcpCommand, + cwd: String, + env: HashMap, + sandbox: Arc, + state: TransportState, +} + +impl SandboxAcpTransport { + pub(crate) fn new( + command: AcpCommand, + cwd: String, + env: HashMap, + sandbox: Arc, + state: TransportState, + ) -> Self { + Self { + command, + cwd, + env, + sandbox, + state, + } + } +} + +impl ConnectTo for SandboxAcpTransport { + async fn connect_to(self, client: impl ConnectTo) -> AcpProtocolResult<()> { + let mut env = self.command.env().clone(); + env.extend(self.env); + + let process = match self + .sandbox + .spawn_stdio_process( + &self.command.to_shell_command(), + Some(&self.cwd), + Some(&env), + None, + ) + .await + { + Ok(process) => process, + Err(error) => { + self.state.set_startup_error(error).await; + return Err(internal_error("ACP process failed to start")); + } + }; + + let handle = process.handle.clone(); + let stderr = process.stderr.clone(); + self.state.set_process(handle.clone(), stderr.clone()).await; + + let incoming_lines = Box::pin(BufReader::new(process.stdout.compat()).lines()) + as Pin> + Send>>; + let outgoing_sink = Box::pin(unfold( + process.stdin.compat_write(), + async move |mut writer, line: String| { + let mut bytes = line.into_bytes(); + bytes.push(b'\n'); + writer.write_all(&bytes).await?; + Ok::<_, std::io::Error>(writer) + }, + )); + + let protocol = agent_client_protocol::ConnectTo::::connect_to( + Lines::new(outgoing_sink, incoming_lines), + client, + ); + tokio::select! { + result = protocol => { + if let Err(err) = handle.terminate().await { + tracing::warn!(error = %err, "Failed to terminate ACP process after protocol completion"); + } + let _ = timeout(Duration::from_millis(500), handle.wait()).await; + result + } + termination = handle.wait() => { + let termination = termination.map_err(ProtocolError::into_internal_error)?; + let stderr = stderr.tail_string().await; + let exit_code = termination + .exit_code + .map_or_else(|| "unknown".to_string(), |code| code.to_string()); + Err(internal_error(format!( + "ACP process exited before protocol completed: termination={}, exit_code={exit_code}, stderr={stderr}", + termination.termination, + ))) + } + } + } +} diff --git a/lib/crates/fabro-acp/tests/session.rs b/lib/crates/fabro-acp/tests/session.rs new file mode 100644 index 000000000..435dbfadb --- /dev/null +++ b/lib/crates/fabro-acp/tests/session.rs @@ -0,0 +1,449 @@ +use std::collections::HashMap; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +use agent_client_protocol::schema::StopReason; +use fabro_acp::{AcpError, AcpRunRequest, AcpRunResult, resolve_acp_command, run_acp_turn}; +use fabro_sandbox::test_support::MockSandbox; +use fabro_sandbox::{LocalSandbox, Sandbox, shell_quote}; +use fabro_util::error::collect_chain; +use tokio::fs::{read_to_string, write}; +use tokio::process::Command; +use tokio::sync::Notify; +use tokio::time::{sleep, timeout}; +use tokio_util::sync::CancellationToken; + +const ACP_TEST_TIMEOUT_MS: u64 = 30_000; + +#[allow( + unused, + unreachable_pub, + reason = "integration test imports the shared test fixture source as a private module" +)] +#[path = "../src/test_support.rs"] +mod test_support; + +use test_support::fake_acp_agent_script; + +#[tokio::test] +async fn stdio_spawn_failure_returns_sandbox_error() { + const SANDBOX_FAILURE: &str = "ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet"; + + let command = resolve_acp_command(Some("fake-acp-agent")).expect("resolve ACP command"); + let mut sandbox = MockSandbox::linux(); + sandbox.stdio_process_error = Some(SANDBOX_FAILURE.to_string()); + let sandbox: Arc = Arc::new(sandbox); + + let result = run_acp_turn(AcpRunRequest { + command, + prompt: "hello".to_string(), + cwd: "/workspace".to_string(), + timeout_ms: Some(ACP_TEST_TIMEOUT_MS), + env: HashMap::new(), + sandbox, + cancel_token: CancellationToken::new(), + on_activity: None, + }) + .await; + let Err(error) = result else { + panic!("stdio spawn failure should fail"); + }; + + assert!( + matches!(error, AcpError::Sandbox(_)), + "expected sandbox error, got {error:?}" + ); + let chain = collect_chain(&error); + assert!( + chain.iter().any(|cause| cause == SANDBOX_FAILURE), + "cause chain should contain sandbox failure, got: {chain:?}" + ); +} + +#[tokio::test] +async fn session_lifecycle_initializes_sends_prompt_and_aggregates_text() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let script_path = tempdir.path().join("fake_acp_agent.py"); + let record_path = tempdir.path().join("methods.txt"); + write(&script_path, fake_acp_agent_script()) + .await + .expect("write fake ACP agent"); + + let raw_command = format!("python3 {}", shell_quote(&script_path.to_string_lossy())); + let command = resolve_acp_command(Some(&raw_command)).expect("resolve ACP command"); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + + let result = run_acp_turn(AcpRunRequest { + command, + prompt: "hello".to_string(), + cwd: tempdir.path().to_string_lossy().into_owned(), + timeout_ms: Some(ACP_TEST_TIMEOUT_MS), + env: HashMap::from([( + "ACP_RECORD".to_string(), + record_path.to_string_lossy().into_owned(), + )]), + sandbox, + cancel_token: CancellationToken::new(), + on_activity: None, + }) + .await + .expect("run ACP turn"); + + assert_eq!(result.text, "hello from acp"); + assert_eq!(result.stop_reason, StopReason::EndTurn); + assert_eq!( + read_to_string(record_path) + .await + .expect("read method record"), + "initialize\nsession/new\nsession/prompt\n" + ); +} + +#[tokio::test] +async fn permission_request_selects_allow_always() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let permission_path = tempdir.path().join("permission.json"); + + let result = run_fake_agent( + tempdir.path(), + HashMap::from([ + ("ACP_MODE".to_string(), "permission".to_string()), + ( + "ACP_PERMISSION".to_string(), + permission_path.to_string_lossy().into_owned(), + ), + ]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect("run ACP turn"); + + assert_eq!(result.text, "hello from acp"); + let permission = read_to_string(permission_path) + .await + .expect("read permission record"); + assert!(permission.contains(r#""outcome":"selected""#)); + assert!(permission.contains(r#""optionId":"always""#)); +} + +#[tokio::test] +async fn runs_inside_sandbox_and_uses_requested_cwd() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let cwd_path = tempdir.path().join("session_new.json"); + + let result = run_fake_agent( + tempdir.path(), + HashMap::from([ + ("ACP_MODE".to_string(), "write_file".to_string()), + ( + "ACP_SESSION_NEW_PARAMS".to_string(), + cwd_path.to_string_lossy().into_owned(), + ), + ]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect("run ACP turn"); + + assert_eq!(result.text, "hello from acp"); + assert_eq!( + read_to_string(tempdir.path().join("hello.txt")) + .await + .expect("read sandbox output file"), + "hello from sandbox\n" + ); + assert!( + read_to_string(cwd_path) + .await + .expect("read session/new params") + .contains(&tempdir.path().to_string_lossy().into_owned()) + ); +} + +#[tokio::test] +async fn cancellation_sends_session_cancel_and_returns_cancelled() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let cancel_path = tempdir.path().join("cancel.txt"); + let tempdir_path = tempdir.path().to_path_buf(); + let cancel_path_for_task = cancel_path.clone(); + let cancel_token = CancellationToken::new(); + let cancel_for_task = cancel_token.clone(); + let prompt_started = Arc::new(Notify::new()); + let prompt_started_for_task = prompt_started.clone(); + + let task = tokio::spawn(async move { + run_fake_agent_with_activity( + &tempdir_path, + HashMap::from([ + ("ACP_MODE".to_string(), "cancel".to_string()), + ( + "ACP_CANCEL_RECORD".to_string(), + cancel_path_for_task.to_string_lossy().into_owned(), + ), + ]), + Some(ACP_TEST_TIMEOUT_MS), + cancel_for_task, + Some(Arc::new(move || prompt_started_for_task.notify_one())), + ) + .await + }); + + timeout( + Duration::from_millis(ACP_TEST_TIMEOUT_MS), + prompt_started.notified(), + ) + .await + .expect("fake ACP agent should acknowledge session/prompt before cancellation"); + cancel_token.cancel(); + let err = task + .await + .expect("join cancellation task") + .expect_err("cancelled turn should error"); + + assert!(matches!(err, AcpError::Cancelled)); + assert_eq!( + read_to_string(cancel_path) + .await + .expect("read cancel record"), + "session/cancel\n" + ); +} + +#[tokio::test] +async fn pre_session_cancellation_returns_cancelled() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let cancel_token = CancellationToken::new(); + cancel_token.cancel(); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_MODE".to_string(), "slow_initialize".to_string())]), + Some(1_000), + cancel_token, + ) + .await + .expect_err("pre-session cancellation should error"); + + assert!(matches!(err, AcpError::Cancelled)); +} + +#[tokio::test] +async fn successful_turn_terminates_lingering_agent_process() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + let pid_path = tempdir.path().join("agent.pid"); + + let result = run_fake_agent( + tempdir.path(), + HashMap::from([ + ("ACP_MODE".to_string(), "linger_after_response".to_string()), + ( + "ACP_PID_RECORD".to_string(), + pid_path.to_string_lossy().into_owned(), + ), + ]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect("run ACP turn"); + + sleep(Duration::from_millis(100)).await; + let pid = read_to_string(&pid_path).await.expect("read agent pid"); + let still_running = process_is_running(pid.trim()).await; + if still_running { + let _ = Command::new("kill") + .arg("-TERM") + .arg(pid.trim()) + .status() + .await; + } + + assert_eq!(result.text, "hello from acp"); + assert!( + !still_running, + "successful ACP turn should not leave lingering agent process" + ); +} + +#[tokio::test] +async fn refusal_stop_reason_returns_text() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let result = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_STOP_REASON".to_string(), "refusal".to_string())]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect("run ACP turn"); + + assert_eq!(result.text, "hello from acp"); + assert_eq!(result.stop_reason, StopReason::Refusal); +} + +#[tokio::test] +async fn max_tokens_stop_reason_returns_partial_text_error() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_STOP_REASON".to_string(), "max_tokens".to_string())]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect_err("max_tokens should return stop reason error"); + + let AcpError::StopReason { stop_reason, text } = err else { + panic!("expected stop reason error"); + }; + assert_eq!(stop_reason, "max_tokens"); + assert_eq!(text, "hello from acp"); +} + +#[tokio::test] +async fn max_turn_requests_stop_reason_returns_partial_text_error() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([( + "ACP_STOP_REASON".to_string(), + "max_turn_requests".to_string(), + )]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect_err("max_turn_requests should return stop reason error"); + + let AcpError::StopReason { stop_reason, text } = err else { + panic!("expected stop reason error"); + }; + assert_eq!(stop_reason, "max_turn_requests"); + assert_eq!(text, "hello from acp"); +} + +#[tokio::test] +async fn timeout_terminates_process_and_returns_timeout() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_MODE".to_string(), "timeout".to_string())]), + Some(100), + CancellationToken::new(), + ) + .await + .expect_err("timeout should error"); + + assert!(matches!(err, AcpError::TimedOut { .. })); +} + +#[tokio::test] +async fn malformed_json_returns_protocol_error() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_MODE".to_string(), "malformed".to_string())]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect_err("malformed JSON should error"); + + assert!(matches!(err, AcpError::Protocol(_))); +} + +#[tokio::test] +async fn early_exit_returns_protocol_error_with_stderr() { + let tempdir = tempfile::tempdir().expect("create tempdir"); + + let err = run_fake_agent( + tempdir.path(), + HashMap::from([("ACP_MODE".to_string(), "early_exit".to_string())]), + Some(ACP_TEST_TIMEOUT_MS), + CancellationToken::new(), + ) + .await + .expect_err("early exit should error"); + + let AcpError::Protocol(error) = err else { + panic!("expected protocol error"); + }; + let message = error.to_string(); + assert!( + message.contains("exit_code=2"), + "early exit should include exit code in diagnostic: {message}" + ); + assert!( + message.contains("early boom"), + "early exit should include stderr tail in diagnostic: {message}" + ); +} + +async fn run_fake_agent( + tempdir: &Path, + env: HashMap, + timeout_ms: Option, + cancel_token: CancellationToken, +) -> Result { + run_fake_agent_with_activity(tempdir, env, timeout_ms, cancel_token, None).await +} + +async fn run_fake_agent_with_activity( + tempdir: &Path, + env: HashMap, + timeout_ms: Option, + cancel_token: CancellationToken, + on_activity: Option>, +) -> Result { + let script_path = tempdir.join("fake_acp_agent.py"); + write(&script_path, fake_acp_agent_script()) + .await + .expect("write fake ACP agent"); + let raw_command = format!("python3 {}", shell_quote(&script_path.to_string_lossy())); + let command = resolve_acp_command(Some(&raw_command)).expect("resolve ACP command"); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.to_path_buf())); + + run_acp_turn(AcpRunRequest { + command, + prompt: "hello".to_string(), + cwd: tempdir.to_string_lossy().into_owned(), + timeout_ms, + env, + sandbox, + cancel_token, + on_activity, + }) + .await +} + +async fn process_is_running(pid: &str) -> bool { + let Ok(status) = Command::new("kill").arg("-0").arg(pid).status().await else { + return false; + }; + if !status.success() { + return false; + } + + let Ok(output) = Command::new("ps") + .args(["-ww", "-o", "stat=", "-p", pid]) + .output() + .await + else { + return true; + }; + if !output.status.success() { + return false; + } + String::from_utf8_lossy(&output.stdout) + .chars() + .find(|ch| !ch.is_whitespace()) + .is_none_or(|state| !matches!(state, 'Z' | 'z')) +} diff --git a/lib/crates/fabro-agent/src/lib.rs b/lib/crates/fabro-agent/src/lib.rs index 9f75a68e7..db293eda5 100644 --- a/lib/crates/fabro-agent/src/lib.rs +++ b/lib/crates/fabro-agent/src/lib.rs @@ -41,8 +41,9 @@ pub use profiles::{AnthropicProfile, EnvContext, GeminiProfile, OpenAiProfile}; pub use read_before_write_sandbox::ReadBeforeWriteSandbox; pub use sandbox::{ CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox, - SandboxEvent, SandboxEventCallback, WorktreeEvent, WorktreeEventCallback, WorktreeOptions, - WorktreeSandbox, format_lines_numbered, shell_quote, + SandboxEvent, SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle, + WorktreeEvent, WorktreeEventCallback, WorktreeOptions, WorktreeSandbox, format_lines_numbered, + shell_quote, }; pub use session::{ CompletionCoordinator, Session, SessionControlHandle, StaticEnvProvider, SteeringItem, diff --git a/lib/crates/fabro-agent/src/mcp_integration.rs b/lib/crates/fabro-agent/src/mcp_integration.rs index ed4ef4574..4c1514d5c 100644 --- a/lib/crates/fabro-agent/src/mcp_integration.rs +++ b/lib/crates/fabro-agent/src/mcp_integration.rs @@ -62,6 +62,8 @@ mod tests { command: vec!["python3".into(), test_server], env: HashMap::new(), }, + current_dir: None, + clear_env: false, startup_timeout_secs: 10, tool_timeout_secs: 30, } diff --git a/lib/crates/fabro-agent/src/sandbox.rs b/lib/crates/fabro-agent/src/sandbox.rs index 0a7335364..6256967e0 100644 --- a/lib/crates/fabro-agent/src/sandbox.rs +++ b/lib/crates/fabro-agent/src/sandbox.rs @@ -3,6 +3,7 @@ // `crate::delegate_sandbox!` invocations continue to work. pub use fabro_sandbox::{ CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox, - SandboxEvent, SandboxEventCallback, WorktreeEvent, WorktreeEventCallback, WorktreeOptions, + SandboxEvent, SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle, + StdioProcessTermination, WorktreeEvent, WorktreeEventCallback, WorktreeOptions, WorktreeSandbox, delegate_sandbox, format_lines_numbered, shell_quote, }; diff --git a/lib/crates/fabro-agent/src/session.rs b/lib/crates/fabro-agent/src/session.rs index 045038a63..143898983 100644 --- a/lib/crates/fabro-agent/src/session.rs +++ b/lib/crates/fabro-agent/src/session.rs @@ -493,6 +493,8 @@ impl Session { resolved.push(McpServerSettings { name: config.name.clone(), transport: McpTransport::Http { url, headers }, + current_dir: config.current_dir.clone(), + clear_env: config.clear_env, startup_timeout_secs: config.startup_timeout_secs, tool_timeout_secs: config.tool_timeout_secs, }); @@ -3367,6 +3369,8 @@ mod tests { command: vec!["python3".into(), test_server], env: HashMap::new(), }, + current_dir: None, + clear_env: false, startup_timeout_secs: 10, tool_timeout_secs: 30, }], diff --git a/lib/crates/fabro-cli/Cargo.toml b/lib/crates/fabro-cli/Cargo.toml index d3f4c94d2..3765e6ef8 100644 --- a/lib/crates/fabro-cli/Cargo.toml +++ b/lib/crates/fabro-cli/Cargo.toml @@ -31,6 +31,8 @@ fabro-hooks = { path = "../fabro-hooks" } fabro-install = { path = "../fabro-install" } fabro-interview = { path = "../fabro-interview" } fabro-mcp = { path = "../fabro-mcp" } +fabro-mcp-server = { path = "../fabro-mcp-server" } +fabro-manifest = { path = "../fabro-manifest" } fabro-proc = { path = "../fabro-proc" } fabro-sandbox = { path = "../fabro-sandbox", features = ["daytona"] } fabro-checkpoint = { path = "../fabro-checkpoint" } @@ -113,6 +115,7 @@ chrono = { workspace = true } [dev-dependencies] assert_cmd = "2" +fabro-acp = { path = "../fabro-acp", features = ["test-support"] } fabro-build-support = { path = "../build-support" } fabro-server = { path = "../fabro-server", features = ["test-support"] } insta = { workspace = true, features = ["filters"] } diff --git a/lib/crates/fabro-cli/src/args.rs b/lib/crates/fabro-cli/src/args.rs index 8525fa0df..0ed502b1b 100644 --- a/lib/crates/fabro-cli/src/args.rs +++ b/lib/crates/fabro-cli/src/args.rs @@ -168,6 +168,49 @@ pub(crate) struct ServerConnectionArgs { pub(crate) target: ServerTargetArgs, } +#[derive(Args)] +pub(crate) struct McpNamespace { + #[command(subcommand)] + pub(crate) command: McpCommand, +} + +#[derive(Subcommand)] +pub(crate) enum McpCommand { + /// Start the Fabro MCP server over stdio + Start(McpStartArgs), + /// Print MCP client configuration JSON + Config(McpConfigArgs), + /// Configure an MCP client to launch Fabro + Init(McpInitArgs), +} + +#[derive(Args, Debug, Clone, Default)] +pub(crate) struct McpStartArgs { + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Args, Debug, Clone, Default)] +pub(crate) struct McpConfigArgs { + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Args, Debug, Clone)] +pub(crate) struct McpInitArgs { + pub(crate) agent: McpAgent, + + #[command(flatten)] + pub(crate) connection: ServerConnectionArgs, +} + +#[derive(Debug, Clone, Copy, ValueEnum)] +pub(crate) enum McpAgent { + Claude, + Cursor, + Windsurf, +} + #[derive(Args, Debug, Clone, Default)] pub(crate) struct InputOverrideArgs { /// Override a workflow input value (repeatable, format: KEY=VALUE) @@ -1118,6 +1161,8 @@ pub(crate) enum Commands { #[command(subcommand)] command: Option, }, + /// Model Context Protocol server + Mcp(McpNamespace), /// Server operations Server(ServerNamespace), /// Check environment and integration health @@ -1209,6 +1254,11 @@ impl Commands { Some(ModelsCommand::Test(_)) => "model test", None => "model", }, + Self::Mcp(ns) => match &ns.command { + McpCommand::Start(_) => "mcp start", + McpCommand::Config(_) => "mcp config", + McpCommand::Init(_) => "mcp init", + }, Self::Server(ns) => match &ns.command { ServerCommand::Start(_) => "server start", ServerCommand::Stop(_) => "server stop", diff --git a/lib/crates/fabro-cli/src/command_context.rs b/lib/crates/fabro-cli/src/command_context.rs index 34b7d3876..5403c57ce 100644 --- a/lib/crates/fabro-cli/src/command_context.rs +++ b/lib/crates/fabro-cli/src/command_context.rs @@ -102,6 +102,10 @@ impl CommandContext { &self.cwd } + pub(crate) fn storage_dir(&self) -> &Path { + &self.storage_dir + } + pub(crate) fn run_settings(&self) -> Result<&RunNamespace> { self.run_settings .as_ref() diff --git a/lib/crates/fabro-cli/src/commands/graph.rs b/lib/crates/fabro-cli/src/commands/graph.rs index dbfae41fc..1c046d311 100644 --- a/lib/crates/fabro-cli/src/commands/graph.rs +++ b/lib/crates/fabro-cli/src/commands/graph.rs @@ -12,13 +12,13 @@ use std::io::Write; use anyhow::{Context, bail}; use fabro_api::types; use fabro_config::user::active_settings_path; +use fabro_manifest::{ManifestBuildInput, build_run_manifest}; use fabro_util::terminal::Styles; use tracing::debug; use crate::args::{GraphArgs, GraphDirection, GraphOutputFormat}; use crate::command_context::CommandContext; use crate::commands::run::output::api_diagnostics_to_local; -use crate::manifest_builder::{ManifestBuildInput, build_run_manifest}; use crate::shared::{absolute_or_current, print_diagnostics, print_json_pretty, relative_path}; pub(crate) async fn run( diff --git a/lib/crates/fabro-cli/src/commands/mcp/mod.rs b/lib/crates/fabro-cli/src/commands/mcp/mod.rs new file mode 100644 index 000000000..f6a64a0ca --- /dev/null +++ b/lib/crates/fabro-cli/src/commands/mcp/mod.rs @@ -0,0 +1,91 @@ +use std::fmt::Write as _; + +use anyhow::{Context as _, Result}; + +use crate::args::{McpAgent, McpCommand, McpNamespace, ServerConnectionArgs}; +use crate::command_context::CommandContext; +use crate::server_client; + +pub(crate) async fn dispatch(ns: McpNamespace, base_ctx: &CommandContext) -> Result<()> { + match ns.command { + McpCommand::Start(args) => { + fabro_mcp_server::start(server_settings(base_ctx, &args.connection)?).await + } + McpCommand::Config(args) => { + let json = fabro_mcp_server::config_json(&config_settings(&args.connection))?; + let _ = write!(base_ctx.printer().stdout_important(), "{json}"); + Ok(()) + } + McpCommand::Init(args) => { + fabro_mcp_server::init_agent(&init_settings(args.agent, &args.connection)?)?; + Ok(()) + } + } +} + +fn server_settings( + base_ctx: &CommandContext, + connection: &ServerConnectionArgs, +) -> Result { + let connection_ctx = base_ctx.with_connection(connection)?; + let target = connection.target.clone(); + let user_settings = connection_ctx.user_settings().clone(); + let storage_dir = connection_ctx.storage_dir().to_path_buf(); + let base_config_path = connection_ctx.base_config_path().to_path_buf(); + let config_path = base_config_path.clone(); + let client_factory: fabro_mcp_server::FabroClientFactory = std::sync::Arc::new(move || { + let target = target.clone(); + let user_settings = user_settings.clone(); + let storage_dir = storage_dir.clone(); + let base_config_path = base_config_path.clone(); + let future: fabro_mcp_server::FabroClientFuture = Box::pin(async move { + server_client::connect_server_with_settings( + &target, + &user_settings, + &storage_dir, + &base_config_path, + ) + .await + }); + future + }); + Ok(fabro_mcp_server::FabroMcpServerSettings { + client_factory, + config_path, + cwd: base_ctx.cwd().to_path_buf(), + }) +} + +fn init_settings( + agent: McpAgent, + connection: &ServerConnectionArgs, +) -> Result { + Ok(fabro_mcp_server::McpInitSettings { + agent: McpAgentForServer(agent).into(), + config: config_settings(connection), + home_dir: home_dir()?, + }) +} + +fn config_settings(connection: &ServerConnectionArgs) -> fabro_mcp_server::McpConfigSettings { + fabro_mcp_server::McpConfigSettings { + server: connection.target.server.clone(), + storage_dir: connection.storage_dir.clone_path(), + } +} + +fn home_dir() -> Result { + dirs::home_dir().context("failed to resolve home directory for MCP config") +} + +struct McpAgentForServer(McpAgent); + +impl From for fabro_mcp_server::McpAgent { + fn from(value: McpAgentForServer) -> Self { + match value.0 { + McpAgent::Claude => Self::Claude, + McpAgent::Cursor => Self::Cursor, + McpAgent::Windsurf => Self::Windsurf, + } + } +} diff --git a/lib/crates/fabro-cli/src/commands/mod.rs b/lib/crates/fabro-cli/src/commands/mod.rs index 5f2b24b16..7e5c3ee69 100644 --- a/lib/crates/fabro-cli/src/commands/mod.rs +++ b/lib/crates/fabro-cli/src/commands/mod.rs @@ -7,6 +7,7 @@ pub(crate) mod dump; pub(crate) mod exec; pub(crate) mod graph; pub(crate) mod install; +pub(crate) mod mcp; pub(crate) mod model; pub(crate) mod parse; pub(crate) mod pr; diff --git a/lib/crates/fabro-cli/src/commands/preflight.rs b/lib/crates/fabro-cli/src/commands/preflight.rs index a3d1dbdf6..f51c508ad 100644 --- a/lib/crates/fabro-cli/src/commands/preflight.rs +++ b/lib/crates/fabro-cli/src/commands/preflight.rs @@ -1,5 +1,6 @@ use anyhow::bail; use fabro_config::user::active_settings_path; +use fabro_manifest::{ManifestBuildInput, build_run_manifest}; use fabro_util::terminal::Styles; use crate::args::PreflightArgs; @@ -8,7 +9,7 @@ use crate::commands::run::output::{ api_check_report_to_local, api_diagnostics_to_local, print_workflow_summary, }; use crate::commands::run::overrides::preflight_args_overrides; -use crate::manifest_builder::{ManifestBuildInput, build_run_manifest, preflight_manifest_args}; +use crate::manifest_args::preflight_manifest_args; use crate::shared::{cyan_spinner, print_json_pretty}; pub(crate) async fn execute( diff --git a/lib/crates/fabro-cli/src/commands/run/create.rs b/lib/crates/fabro-cli/src/commands/run/create.rs index c56a89024..bcbac4a8c 100644 --- a/lib/crates/fabro-cli/src/commands/run/create.rs +++ b/lib/crates/fabro-cli/src/commands/run/create.rs @@ -1,6 +1,7 @@ use anyhow::{Context as _, bail}; use fabro_config::RunLayer; use fabro_config::user::active_settings_path; +use fabro_manifest::{ManifestBuildInput, build_run_manifest}; use fabro_server::manifest_validation; use fabro_types::RunId; use fabro_util::terminal::Styles; @@ -9,7 +10,7 @@ use super::output::{api_diagnostics_to_local, print_workflow_summary}; use super::overrides::run_args_overrides; use crate::args::RunArgs; use crate::command_context::CommandContext; -use crate::manifest_builder::{ManifestBuildInput, build_run_manifest, run_manifest_args}; +use crate::manifest_args::run_manifest_args; pub(crate) struct CreatedRun { pub(crate) run_id: RunId, diff --git a/lib/crates/fabro-cli/src/commands/run/overrides.rs b/lib/crates/fabro-cli/src/commands/run/overrides.rs index 47af597c7..d4de0183f 100644 --- a/lib/crates/fabro-cli/src/commands/run/overrides.rs +++ b/lib/crates/fabro-cli/src/commands/run/overrides.rs @@ -2,14 +2,11 @@ use std::collections::HashMap; use std::path::{Path, PathBuf}; use anyhow::{Result, anyhow}; -use fabro_config::{ - CliLayer, CliOutputLayer, ReplaceMap, RunExecutionLayer, RunGoalLayer, RunLayer, RunModelLayer, - RunSandboxLayer, parse_input_overrides, -}; +use fabro_config::{CliLayer, CliOutputLayer, RunGoalLayer, RunLayer, parse_input_overrides}; +use fabro_manifest::{RunOverrideInput, build_run_overrides}; use fabro_sandbox::SandboxProvider; use fabro_types::settings::cli::OutputVerbosity; use fabro_types::settings::interp::InterpString; -use fabro_types::settings::run::{ApprovalMode, RunMode}; use crate::args::{PreflightArgs, RunArgs}; @@ -32,48 +29,6 @@ pub(crate) fn parse_labels(labels: &[String]) -> HashMap { .collect() } -fn model_from_args(model: Option<&str>, provider: Option<&str>) -> Option { - if model.is_none() && provider.is_none() { - return None; - } - Some(RunModelLayer { - provider: provider.map(InterpString::parse), - name: model.map(InterpString::parse), - fallbacks: Vec::new(), - controls: None, - }) -} - -fn sandbox_layer( - sandbox: Option, - preserve: Option, -) -> Option { - if sandbox.is_none() && preserve.is_none() { - return None; - } - Some(RunSandboxLayer { - provider: sandbox.map(|p| p.to_string()), - preserve, - ..RunSandboxLayer::default() - }) -} - -fn execution_layer(dry_run: Option, auto_approve: Option) -> Option { - if dry_run.is_none() && auto_approve.is_none() { - return None; - } - Some(RunExecutionLayer { - mode: dry_run.map(|d| if d { RunMode::DryRun } else { RunMode::Normal }), - approval: auto_approve.map(|a| { - if a { - ApprovalMode::Auto - } else { - ApprovalMode::Prompt - } - }), - }) -} - fn cli_layer_for_verbose(verbose: bool) -> Option { verbose.then(|| CliLayer { output: Some(CliOutputLayer { @@ -120,24 +75,22 @@ fn current_dir_or_dot() -> PathBuf { } pub(crate) fn run_args_overrides(args: &RunArgs) -> Result { - let model = model_from_args(args.model.as_deref(), args.provider.as_deref()); - let sandbox = sandbox_layer( - args.sandbox.map(Into::into), - sparse_flag(args.preserve_sandbox), - ); - let execution = execution_layer(sparse_flag(args.dry_run), sparse_flag(args.auto_approve)); - let cwd = current_dir_or_dot(); let goal = goal_layer_from_args(args.goal.as_deref(), args.goal_file.as_deref(), &cwd)?; - - let run = RunLayer { - goal, - metadata: ReplaceMap::from(parse_labels(&args.label)), - model, - sandbox, - execution, - ..RunLayer::default() - }; + let sandbox = args.sandbox.map(SandboxProvider::from); + let sandbox_provider = sandbox.as_ref().map(ToString::to_string); + let mut run = build_run_overrides(RunOverrideInput { + goal: None, + model: args.model.as_deref(), + provider: args.provider.as_deref(), + sandbox: sandbox_provider.as_deref(), + docker_image: None, + preserve_sandbox: sparse_flag(args.preserve_sandbox), + dry_run: sparse_flag(args.dry_run), + auto_approve: sparse_flag(args.auto_approve), + labels: parse_labels(&args.label), + }); + run.goal = goal; Ok(ManifestSettingsOverrides { run: Some(run), @@ -147,21 +100,23 @@ pub(crate) fn run_args_overrides(args: &RunArgs) -> Result Result { - let model = model_from_args(args.model.as_deref(), args.provider.as_deref()); - let sandbox = args.sandbox.map(|s| RunSandboxLayer { - provider: Some(SandboxProvider::from(s).to_string()), - ..RunSandboxLayer::default() - }); - let cwd = current_dir_or_dot(); let goal = goal_layer_from_args(args.goal.as_deref(), args.goal_file.as_deref(), &cwd)?; - - let run = RunLayer { - goal, - model, - sandbox, - ..RunLayer::default() - }; + let sandbox_provider = args + .sandbox + .map(|sandbox| SandboxProvider::from(sandbox).to_string()); + let mut run = build_run_overrides(RunOverrideInput { + goal: None, + model: args.model.as_deref(), + provider: args.provider.as_deref(), + sandbox: sandbox_provider.as_deref(), + docker_image: None, + preserve_sandbox: None, + dry_run: None, + auto_approve: None, + labels: HashMap::new(), + }); + run.goal = goal; Ok(ManifestSettingsOverrides { run: Some(run), diff --git a/lib/crates/fabro-cli/src/commands/run/run_progress/mod.rs b/lib/crates/fabro-cli/src/commands/run/run_progress/mod.rs index a4d7d1b46..5eb37ff79 100644 --- a/lib/crates/fabro-cli/src/commands/run/run_progress/mod.rs +++ b/lib/crates/fabro-cli/src/commands/run/run_progress/mod.rs @@ -472,6 +472,7 @@ mod tests { use fabro_agent::{AgentEvent, SandboxEvent}; use fabro_llm::types::TokenCounts; use fabro_model::{ModelRef, Provider}; + use fabro_types::run_event::CliEnsureCompletedProps; use fabro_types::{ MetadataSnapshotFailureKind, MetadataSnapshotPhase, ParallelBranchId, SandboxProvider, StageId, fixtures, @@ -527,6 +528,24 @@ mod tests { ui.handle_event(&stored); } + fn emit_body(ui: &mut ProgressUI, body: fabro_types::EventBody) { + ui.handle_event(&RunEvent { + id: "evt_legacy".to_string(), + ts: Utc::now(), + run_id: fixtures::RUN_1, + node_id: None, + node_label: None, + stage_id: None, + parallel_group_id: None, + parallel_branch_id: None, + session_id: None, + parent_session_id: None, + tool_call_id: None, + actor: None, + body, + }); + } + fn agent_event(stage: &str, event: AgentEvent) -> Event { Event::Agent { stage: stage.into(), @@ -860,13 +879,16 @@ mod tests { }); emit(&mut ui, Event::SetupStarted { command_count: 2 }); emit(&mut ui, Event::SetupCompleted { duration_ms: 8200 }); - emit(&mut ui, Event::CliEnsureCompleted { - cli_name: "gh".into(), - provider: "github".into(), - already_installed: false, - node_installed: false, - duration_ms: 600, - }); + emit_body( + &mut ui, + fabro_types::EventBody::CliEnsureCompleted(CliEnsureCompletedProps { + cli_name: "gh".into(), + provider: "github".into(), + already_installed: false, + node_installed: false, + duration_ms: 600, + }), + ); emit(&mut ui, Event::DevcontainerResolved { dockerfile_lines: 24, environment_count: 3, diff --git a/lib/crates/fabro-cli/src/commands/run/start.rs b/lib/crates/fabro-cli/src/commands/run/start.rs index 0af737d81..5ad57539b 100644 --- a/lib/crates/fabro-cli/src/commands/run/start.rs +++ b/lib/crates/fabro-cli/src/commands/run/start.rs @@ -8,5 +8,5 @@ pub(crate) async fn start_run_with_client( run_id: &RunId, resume: bool, ) -> Result<()> { - client.start_run(run_id, resume).await + client.start_run(run_id, resume).await.map(|_| ()) } diff --git a/lib/crates/fabro-cli/src/commands/runs/archive.rs b/lib/crates/fabro-cli/src/commands/runs/archive.rs index 9694bd829..38a5308fe 100644 --- a/lib/crates/fabro-cli/src/commands/runs/archive.rs +++ b/lib/crates/fabro-cli/src/commands/runs/archive.rs @@ -71,7 +71,7 @@ async fn run_bulk(action: Action, identifiers: &[String], ctx: &CommandContext) Action::Unarchive => client.unarchive_run(&run_id).await, }; match result { - Ok(()) => { + Ok(_) => { let run_id_string = run_id.to_string(); changed.push(run_id_string.clone()); if !json { diff --git a/lib/crates/fabro-cli/src/commands/validate.rs b/lib/crates/fabro-cli/src/commands/validate.rs index 74495fadf..d607c4b2c 100644 --- a/lib/crates/fabro-cli/src/commands/validate.rs +++ b/lib/crates/fabro-cli/src/commands/validate.rs @@ -1,13 +1,13 @@ use anyhow::bail; use fabro_config::RunLayer; use fabro_config::user::active_settings_path; +use fabro_manifest::{ManifestBuildInput, build_run_manifest}; use fabro_server::manifest_validation; use fabro_util::terminal::Styles; use crate::args::ValidateArgs; use crate::command_context::CommandContext; use crate::commands::run::output::api_diagnostics_to_local; -use crate::manifest_builder::{ManifestBuildInput, build_run_manifest}; use crate::shared::{print_diagnostics, print_json_pretty, relative_path}; pub(crate) fn run( diff --git a/lib/crates/fabro-cli/src/lib.rs b/lib/crates/fabro-cli/src/lib.rs deleted file mode 100644 index 34a4cfc3f..000000000 --- a/lib/crates/fabro-cli/src/lib.rs +++ /dev/null @@ -1,9 +0,0 @@ -#![expect( - dead_code, - reason = "the library exports manifest builder helpers while the binary owns most CLI dispatch" -)] - -mod args; -mod manifest_builder; - -pub use manifest_builder::{BuiltManifest, ManifestBuildInput, build_run_manifest}; diff --git a/lib/crates/fabro-cli/src/main.rs b/lib/crates/fabro-cli/src/main.rs index adf4562d6..43f39ad80 100644 --- a/lib/crates/fabro-cli/src/main.rs +++ b/lib/crates/fabro-cli/src/main.rs @@ -10,11 +10,7 @@ mod gh; mod landing; mod local_server; mod logging; -#[allow( - unreachable_pub, - reason = "The library exports manifest builder helpers for tests; the binary includes the same module privately." -)] -mod manifest_builder; +mod manifest_args; mod server_client; mod server_runs; mod shared; @@ -281,6 +277,9 @@ async fn main_inner(worker_token: Option) -> (String, Result<()>) { Commands::Model { command } => { commands::model::execute(command, &base_ctx).await?; } + Commands::Mcp(ns) => { + commands::mcp::dispatch(ns, &base_ctx).await?; + } Commands::Server(ns) => { Box::pin(commands::server::dispatch( ns.command, @@ -1196,7 +1195,7 @@ destination = "{destination}" .expect("should parse"); match *cli.command.unwrap() { Commands::RunCmd(RunCommands::Run(args)) => { - let manifest_args = manifest_builder::run_manifest_args(&args) + let manifest_args = manifest_args::run_manifest_args(&args) .expect("input-only args should be retained"); assert_eq!(manifest_args.input, vec!["foo=bar"]); } diff --git a/lib/crates/fabro-cli/src/manifest_args.rs b/lib/crates/fabro-cli/src/manifest_args.rs new file mode 100644 index 000000000..48fc1dbab --- /dev/null +++ b/lib/crates/fabro-cli/src/manifest_args.rs @@ -0,0 +1,39 @@ +use fabro_api::types; + +use crate::args::{PreflightArgs, RunArgs}; + +pub(crate) fn run_manifest_args(args: &RunArgs) -> Option { + let payload = types::ManifestArgs { + auto_approve: args.auto_approve.then_some(true), + dry_run: args.dry_run.then_some(true), + label: args.label.clone(), + model: args.model.clone(), + preserve_sandbox: args.preserve_sandbox.then_some(true), + provider: args.provider.clone(), + sandbox: args + .sandbox + .map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()), + docker_image: None, + input: args.inputs.values.clone(), + verbose: args.verbose.then_some(true), + }; + (!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload) +} + +pub(crate) fn preflight_manifest_args(args: &PreflightArgs) -> Option { + let payload = types::ManifestArgs { + auto_approve: None, + dry_run: None, + label: Vec::new(), + model: args.model.clone(), + preserve_sandbox: None, + provider: args.provider.clone(), + sandbox: args + .sandbox + .map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()), + docker_image: None, + input: args.inputs.values.clone(), + verbose: args.verbose.then_some(true), + }; + (!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload) +} diff --git a/lib/crates/fabro-cli/src/shared/utilities.rs b/lib/crates/fabro-cli/src/shared/utilities.rs index 433886932..7f3c9f20f 100644 --- a/lib/crates/fabro-cli/src/shared/utilities.rs +++ b/lib/crates/fabro-cli/src/shared/utilities.rs @@ -127,18 +127,7 @@ pub(crate) fn color_if(use_color: bool, color: Color) -> Option { } pub(crate) fn run_status_kind(status: RunStatus) -> &'static str { - match status { - RunStatus::Submitted => "submitted", - RunStatus::Queued => "queued", - RunStatus::Starting => "starting", - RunStatus::Running => "running", - RunStatus::Blocked { .. } => "blocked", - RunStatus::Paused { .. } => "paused", - RunStatus::Removing => "removing", - RunStatus::Succeeded { .. } => "succeeded", - RunStatus::Failed { .. } => "failed", - RunStatus::Dead => "dead", - } + status.kind().into() } pub(crate) fn split_run_path(s: &str) -> Option<(&str, &str)> { diff --git a/lib/crates/fabro-cli/tests/it/cmd/doctor.rs b/lib/crates/fabro-cli/tests/it/cmd/doctor.rs index 075a74f98..9c9092b47 100644 --- a/lib/crates/fabro-cli/tests/it/cmd/doctor.rs +++ b/lib/crates/fabro-cli/tests/it/cmd/doctor.rs @@ -5,7 +5,11 @@ use std::process::Output; +use fabro_auth::{AuthCredential, AuthDetails}; +use fabro_config::Storage; +use fabro_model::Provider; use fabro_test::{fabro_snapshot, test_context, twin_openai}; +use fabro_vault::{SecretType, Vault}; async fn run_success_output(mut cmd: assert_cmd::Command) -> Output { tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone()) @@ -13,6 +17,35 @@ async fn run_success_output(mut cmd: assert_cmd::Command) -> Output { .expect("blocking command task should complete") } +fn toml_path(path: &std::path::Path) -> String { + path.display() + .to_string() + .replace('\\', "\\\\") + .replace('"', "\\\"") +} + +fn seed_openai_vault(storage_dir: &std::path::Path, base_url: &str, api_key: &str) { + let mut vault = + Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load"); + vault + .set( + "openai", + &serde_json::to_string(&AuthCredential { + provider: Provider::OpenAi, + details: AuthDetails::ApiKey { + key: api_key.to_string(), + }, + }) + .expect("OpenAI test credential should serialize"), + SecretType::Credential, + None, + ) + .expect("OpenAI credential should store in test vault"); + vault + .set("OPENAI_BASE_URL", base_url, SecretType::Environment, None) + .expect("OpenAI base URL should store in test vault"); +} + #[test] fn help() { let context = test_context!(); @@ -64,9 +97,28 @@ fn live_doctor() { #[fabro_macros::e2e_test(twin)] async fn twin_doctor() { - let context = test_context!(); + let mut context = test_context!(); let twin = twin_openai().await; let namespace = format!("{}::{}", module_path!(), line!()); + let storage_dir = context.temp_dir.join("doctor-server-storage"); + context.write_home( + ".fabro/settings.toml", + format!( + r#"[server.storage] +root = "{}" + +[server.auth] +methods = ["dev-token"] + +[server.integrations.github] +strategy = "app" +"#, + toml_path(&storage_dir) + ), + ); + seed_openai_vault(&storage_dir, &twin.base_url, &namespace); + context.isolated_server(); + let mut cmd = context.doctor(); cmd.arg("--verbose"); cmd.env_clear(); diff --git a/lib/crates/fabro-cli/tests/it/cmd/fabro.rs b/lib/crates/fabro-cli/tests/it/cmd/fabro.rs index ff151778f..6fc7aaff8 100644 --- a/lib/crates/fabro-cli/tests/it/cmd/fabro.rs +++ b/lib/crates/fabro-cli/tests/it/cmd/fabro.rs @@ -33,6 +33,7 @@ fn help() { archive Mark terminal runs as archived (reviewed, no further action needed). Archived runs are hidden from default listings unarchive Restore archived runs to their prior terminal status model List and test LLM models + mcp Model Context Protocol server server Server operations doctor Check environment and integration health version Show client and server version information diff --git a/lib/crates/fabro-cli/tests/it/cmd/mcp.rs b/lib/crates/fabro-cli/tests/it/cmd/mcp.rs new file mode 100644 index 000000000..358372879 --- /dev/null +++ b/lib/crates/fabro-cli/tests/it/cmd/mcp.rs @@ -0,0 +1,2305 @@ +#![expect( + clippy::disallowed_methods, + reason = "integration tests stage MCP config files with sync std::fs" +)] +#![expect( + clippy::disallowed_types, + reason = "raw stdio regression test intentionally uses blocking std pipes outside Tokio" +)] + +use std::collections::HashMap; +use std::io::{BufRead as _, Write as _}; +use std::path::{Path, PathBuf}; +use std::process::Stdio; + +use chrono::{DateTime, Duration as ChronoDuration, Utc}; +use fabro_client::{AuthEntry, AuthStore, DevTokenEntry, OAuthEntry, StoredSubject}; +use fabro_mcp::client::McpClient; +use fabro_mcp::config::{McpServerSettings, McpTransport}; +use fabro_test::{fabro_json_snapshot, fabro_snapshot, test_context}; +use fabro_types::RunId; +use httpmock::Method::{GET, POST}; +use httpmock::MockServer; + +use super::support::{mock_resolved_run, remote_run_summary_json}; +use crate::support::{ + RealAuthHarness, TEST_DEV_TOKEN, run_projection_json, seed_dev_token_auth, unique_run_id, +}; + +#[test] +fn help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "--help"]); + fabro_snapshot!(context.filters(), cmd, @" + success: true + exit_code: 0 + ----- stdout ----- + Model Context Protocol server + + Usage: fabro mcp [OPTIONS] + + Commands: + start Start the Fabro MCP server over stdio + config Print MCP client configuration JSON + init Configure an MCP client to launch Fabro + help Print this message or the help of the given subcommand(s) + + Options: + --json Output as JSON [env: FABRO_JSON=] + --debug Enable DEBUG-level logging (default is INFO) [env: FABRO_DEBUG=] + --no-upgrade-check Disable automatic upgrade check [env: FABRO_NO_UPGRADE_CHECK=true] + --quiet Suppress non-essential output [env: FABRO_QUIET=] + --verbose Enable verbose output [env: FABRO_VERBOSE=] + -h, --help Print help + ----- stderr ----- + "); +} + +#[test] +fn start_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "start", "--help"]); + fabro_snapshot!(context.filters(), cmd, @" + success: true + exit_code: 0 + ----- stdout ----- + Start the Fabro MCP server over stdio + + Usage: fabro mcp start [OPTIONS] + + Options: + --json Output as JSON [env: FABRO_JSON=] + --storage-dir Local storage directory (default: ~/.fabro/storage) [env: FABRO_STORAGE_DIR=] + --debug Enable DEBUG-level logging (default is INFO) [env: FABRO_DEBUG=] + --server Fabro server target: http(s) URL or absolute Unix socket path [env: FABRO_SERVER=] + --no-upgrade-check Disable automatic upgrade check [env: FABRO_NO_UPGRADE_CHECK=true] + --quiet Suppress non-essential output [env: FABRO_QUIET=] + --verbose Enable verbose output [env: FABRO_VERBOSE=] + -h, --help Print help + ----- stderr ----- + "); +} + +#[test] +fn config_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "config", "--help"]); + fabro_snapshot!(context.filters(), cmd, @" + success: true + exit_code: 0 + ----- stdout ----- + Print MCP client configuration JSON + + Usage: fabro mcp config [OPTIONS] + + Options: + --json Output as JSON [env: FABRO_JSON=] + --storage-dir Local storage directory (default: ~/.fabro/storage) [env: FABRO_STORAGE_DIR=] + --debug Enable DEBUG-level logging (default is INFO) [env: FABRO_DEBUG=] + --server Fabro server target: http(s) URL or absolute Unix socket path [env: FABRO_SERVER=] + --no-upgrade-check Disable automatic upgrade check [env: FABRO_NO_UPGRADE_CHECK=true] + --quiet Suppress non-essential output [env: FABRO_QUIET=] + --verbose Enable verbose output [env: FABRO_VERBOSE=] + -h, --help Print help + ----- stderr ----- + "); +} + +#[test] +fn init_help() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "init", "--help"]); + fabro_snapshot!(context.filters(), cmd, @" + success: true + exit_code: 0 + ----- stdout ----- + Configure an MCP client to launch Fabro + + Usage: fabro mcp init [OPTIONS] + + Arguments: + [possible values: claude, cursor, windsurf] + + Options: + --json Output as JSON [env: FABRO_JSON=] + --storage-dir Local storage directory (default: ~/.fabro/storage) [env: FABRO_STORAGE_DIR=] + --debug Enable DEBUG-level logging (default is INFO) [env: FABRO_DEBUG=] + --server Fabro server target: http(s) URL or absolute Unix socket path [env: FABRO_SERVER=] + --no-upgrade-check Disable automatic upgrade check [env: FABRO_NO_UPGRADE_CHECK=true] + --quiet Suppress non-essential output [env: FABRO_QUIET=] + --verbose Enable verbose output [env: FABRO_VERBOSE=] + -h, --help Print help + ----- stderr ----- + "); +} + +#[test] +fn config_prints_generic_mcp_json() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args(["mcp", "config"]); + fabro_snapshot!(context.filters(), cmd, @r#" + success: true + exit_code: 0 + ----- stdout ----- + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + ----- stderr ----- + "#); +} + +#[test] +fn config_preserves_connection_flags() { + let context = test_context!(); + let mut cmd = context.command(); + cmd.args([ + "mcp", + "config", + "--server", + "https://example.test/api/v1", + "--storage-dir", + "/tmp/fabro-mcp-storage", + ]); + fabro_snapshot!(context.filters(), cmd, @r#" + success: true + exit_code: 0 + ----- stdout ----- + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start", + "--server", + "https://example.test/api/v1", + "--storage-dir", + "/tmp/fabro-mcp-storage" + ] + } + } + } + ----- stderr ----- + "#); +} + +#[test] +fn init_cursor_writes_idempotent_config() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "cursor"]) + .assert() + .success(); + context + .command() + .args(["mcp", "init", "cursor"]) + .assert() + .success(); + + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @r#" + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + "#); +} + +#[test] +fn init_claude_writes_desktop_and_code_configs() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "claude"]) + .assert() + .success(); + + let desktop_config: serde_json::Value = serde_json::from_str( + &std::fs::read_to_string(expected_claude_desktop_config_path(&context.home_dir)).unwrap(), + ) + .unwrap(); + fabro_json_snapshot!(context, desktop_config, @r#" + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + "#); + + let code_config: serde_json::Value = serde_json::from_str( + &std::fs::read_to_string(context.home_dir.join(".claude.json")).unwrap(), + ) + .unwrap(); + fabro_json_snapshot!(context, code_config, @r#" + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + "#); +} + +#[test] +fn init_claude_preserves_existing_claude_code_config() { + let context = test_context!(); + let claude_code_path = context.home_dir.join(".claude.json"); + std::fs::write( + &claude_code_path, + r#"{"numStartups":42,"mcpServers":{"other":{"type":"http","url":"https://example.test/mcp"}}}"#, + ) + .unwrap(); + + context + .command() + .args(["mcp", "init", "claude"]) + .assert() + .success(); + + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(&claude_code_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @r#" + { + "numStartups": 42, + "mcpServers": { + "other": { + "type": "http", + "url": "https://example.test/mcp" + }, + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + "#); +} + +#[test] +fn init_windsurf_writes_config() { + let context = test_context!(); + context + .command() + .args(["mcp", "init", "windsurf"]) + .assert() + .success(); + + let config_path = context + .home_dir + .join(".codeium") + .join("windsurf") + .join("mcp_config.json"); + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @r#" + { + "mcpServers": { + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start" + ] + } + } + } + "#); +} + +#[test] +fn init_preserves_existing_servers() { + let context = test_context!(); + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + std::fs::create_dir_all(config_path.parent().unwrap()).unwrap(); + std::fs::write( + &config_path, + r#"{"mcpServers":{"other":{"command":"other","args":["serve"]}},"theme":"dark"}"#, + ) + .unwrap(); + + context + .command() + .args([ + "mcp", + "init", + "cursor", + "--server", + "https://example.test/api/v1", + ]) + .assert() + .success(); + + let config: serde_json::Value = + serde_json::from_str(&std::fs::read_to_string(config_path).unwrap()).unwrap(); + fabro_json_snapshot!(context, config, @r#" + { + "mcpServers": { + "other": { + "command": "other", + "args": [ + "serve" + ] + }, + "fabro": { + "command": "fabro", + "args": [ + "mcp", + "start", + "--server", + "https://example.test/api/v1" + ] + } + }, + "theme": "dark" + } + "#); +} + +#[test] +fn init_invalid_json_fails_without_overwrite() { + let context = test_context!(); + let config_path = context.home_dir.join(".cursor").join("mcp.json"); + std::fs::create_dir_all(config_path.parent().unwrap()).unwrap(); + std::fs::write(&config_path, "{not json").unwrap(); + + let mut cmd = context.command(); + cmd.args(["mcp", "init", "cursor"]); + fabro_snapshot!(context.filters(), cmd, @" + success: false + exit_code: 1 + ----- stdout ----- + ----- stderr ----- + × failed to parse MCP config [HOME_DIR]/.cursor/mcp.json + ╰─▶ key must be a string at line 1 column 2 + "); + assert_eq!(std::fs::read_to_string(config_path).unwrap(), "{not json"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn stdio_server_initializes_and_lists_run_tools() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &[]).await; + + let tools = client.list_tools().await.unwrap(); + let names: Vec<_> = tools.iter().map(|(name, _, _)| name.as_str()).collect(); + assert_eq!(names, vec![ + "fabro_run_create", + "fabro_run_events", + "fabro_run_gather", + "fabro_run_interact", + "fabro_run_search", + ]); + for (name, _, schema) in &tools { + assert!( + schema.is_object(), + "tool should have input schema: {schema}" + ); + let properties = schema + .get("properties") + .and_then(serde_json::Value::as_object) + .expect("tool input schema should have properties"); + for (property, property_schema) in properties { + assert!( + property_schema.is_object(), + "{name}.{property} should use an object JSON Schema, got {property_schema}" + ); + } + } + let interact_schema = tools + .iter() + .find(|(name, _, _)| name == "fabro_run_interact") + .map(|(_, _, schema)| schema) + .expect("fabro_run_interact tool should be listed"); + assert!( + interact_schema + .pointer("/properties/answer") + .is_some_and(serde_json::Value::is_object), + "fabro_run_interact.answer should have an object JSON Schema: {interact_schema}" + ); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[test] +fn stdio_start_writes_only_json_rpc_to_stdout() { + let context = test_context!(); + let fixture = mcp_stdio_fixture(&context, &[]); + let mut cmd = std::process::Command::new(&fixture.command[0]); + cmd.args(&fixture.command[1..]) + .env_clear() + .envs(&fixture.env) + .current_dir(&fixture.current_dir) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + + let mut child = cmd.spawn().unwrap(); + let mut stdin = child.stdin.take().unwrap(); + writeln!( + stdin, + r#"{{"jsonrpc":"2.0","id":1,"method":"initialize","params":{{"protocolVersion":"2025-06-18","capabilities":{{}},"clientInfo":{{"name":"fabro-test","version":"0.0.0"}}}}}}"# + ) + .unwrap(); + + let stdout = child.stdout.take().unwrap(); + let (tx, rx) = std::sync::mpsc::channel(); + std::thread::spawn(move || { + let mut line = String::new(); + let result = std::io::BufReader::new(stdout).read_line(&mut line); + let _ = tx.send(result.map(|_| line)); + }); + + let line = rx + .recv_timeout(std::time::Duration::from_secs(5)) + .expect("initialize response should arrive") + .expect("stdout should be readable"); + let value: serde_json::Value = serde_json::from_str(line.trim()).unwrap(); + assert_eq!(value["jsonrpc"], "2.0"); + + let _ = child.kill(); + let _ = child.wait(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn stdio_startup_and_list_tools_is_fast() { + let context = test_context!(); + let start = std::time::Instant::now(); + let client = spawn_mcp_client(&context, &[]).await; + let tools = client.list_tools().await.unwrap(); + assert_eq!(tools.len(), 5); + assert!(start.elapsed() < std::time::Duration::from_secs(2)); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_create_and_search_manage_real_runs_with_cli_auth() { + let context = test_context!(); + let harness = + RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let workflow = context.install_fixture("simple.fabro"); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let create = call_tool_json( + &client, + "fabro_run_create", + serde_json::json!({ + "runs": [{ + "workflow": workflow, + "dry_run": true, + "auto_approve": true, + "labels": { "source": "mcp-test" } + }] + }), + ) + .await; + let run_id = create["runs"][0]["run_id"].as_str().unwrap().to_string(); + assert_eq!(create["runs"][0]["started"], true); + + let search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ + "run_ids": [run_id], + "labels": { "source": "mcp-test" }, + "first": 10 + }), + ) + .await; + fabro_json_snapshot!(context, normalize_run_search(search), @r#" + { + "runs": [ + { + "run_id": "[RUN_ID]", + "workflow_name": "Simple", + "workflow_slug": "simple", + "status": "queued", + "archived": false, + "created_at": "[TIMESTAMP]", + "started_at": null, + "completed_at": null, + "labels": { + "source": "mcp-test" + }, + "source_directory": "[SOURCE_DIRECTORY]", + "repo_origin_url": null, + "goal_preview": "Run tests and report results", + "goal_truncated": false + } + ], + "next_cursor": null + } + "#); + + client + .shutdown() + .await + .expect("MCP client should shut down"); + harness.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_run_tools_use_default_local_server_without_server_flag() { + let context = test_context!(); + let workflow = context.install_fixture("simple.fabro"); + let client = spawn_mcp_client(&context, &[]).await; + + let create = call_tool_json( + &client, + "fabro_run_create", + serde_json::json!({ + "runs": [{ + "workflow": workflow, + "dry_run": true, + "auto_approve": true, + "labels": { "source": "mcp-default-server-test" }, + "start": false + }] + }), + ) + .await; + let run_id = create["runs"][0]["run_id"].as_str().unwrap(); + let search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "first": 1 }), + ) + .await; + + assert_eq!(search["runs"][0]["run_id"], run_id); + assert_eq!( + search["runs"][0]["labels"]["source"], + "mcp-default-server-test" + ); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_configured_unix_target_auto_spawns_like_cli() { + let mut context = test_context!(); + let isolated = fabro_test::isolated_storage_dir(); + let storage_dir = isolated.path().join("storage"); + let socket_path = isolated.path().join("configured.sock"); + write_mcp_server_settings(&mut context, &storage_dir, Some(&socket_path)); + let workflow = context.install_fixture("simple.fabro"); + + assert!(!socket_path.exists()); + let client = spawn_mcp_client(&context, &[]).await; + let run_id = create_mcp_run(&client, workflow, false).await; + let search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "first": 1 }), + ) + .await; + + assert_eq!(search["runs"][0]["run_id"], run_id); + assert!( + socket_path.exists(), + "configured Unix socket should be auto-spawned" + ); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_explicit_unix_target_auto_spawns_like_cli() { + let mut context = test_context!(); + let isolated = fabro_test::isolated_storage_dir(); + let storage_dir = isolated.path().join("storage"); + let socket_path = isolated.path().join("explicit.sock"); + write_mcp_server_settings(&mut context, &storage_dir, None); + let workflow = context.install_fixture("simple.fabro"); + let storage_arg = storage_dir.display().to_string(); + let socket_arg = socket_path.display().to_string(); + + assert!(!socket_path.exists()); + let client = spawn_mcp_client(&context, &[ + "--storage-dir", + &storage_arg, + "--server", + &socket_arg, + ]) + .await; + let run_id = create_mcp_run(&client, workflow, false).await; + let search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "first": 1 }), + ) + .await; + + assert_eq!(search["runs"][0]["run_id"], run_id); + assert!( + socket_path.exists(), + "explicit Unix socket should be auto-spawned" + ); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_missing_default_settings_reports_configure_first_error_and_stays_alive() { + let home = tempfile::tempdir().unwrap(); + let workspace = tempfile::tempdir().unwrap(); + let mut env = fabro_test::isolated_env(home.path()); + env.insert( + "FABRO_HOME".to_string(), + home.path().join(".fabro").display().to_string(), + ); + let client = spawn_mcp_client_from_fixture(McpStdioFixture { + command: vec![ + env!("CARGO_BIN_EXE_fabro").to_string(), + "mcp".to_string(), + "start".to_string(), + ], + env, + current_dir: workspace.path().to_path_buf(), + }) + .await; + + let error = call_tool_error_text( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": ["missing"], "first": 1 }), + ) + .await; + + assert!( + error.contains("Cannot reach Fabro server: no settings.toml configured."), + "{error}" + ); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_filters_status_dates_and_paginates() { + let context = test_context!(); + let harness = + RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let workflow = context.install_fixture("simple.fabro"); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let first = create_mcp_run(&client, workflow.clone(), false).await; + let second = create_mcp_run(&client, workflow, false).await; + + let page_one = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ + "labels": { "source": "mcp-test" }, + "status": ["submitted"], + "archived": false, + "created_after": "2000-01-01", + "created_before": "2100-01-01T00:00:00Z", + "first": 1 + }), + ) + .await; + let cursor = page_one["next_cursor"] + .as_str() + .expect("first page should have cursor"); + let page_two = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ + "labels": { "source": "mcp-test" }, + "status": ["submitted"], + "archived": false, + "after": cursor, + "first": 1 + }), + ) + .await; + + let page_one_id = page_one["runs"][0]["run_id"].as_str().unwrap(); + let page_two_id = page_two["runs"][0]["run_id"].as_str().unwrap(); + assert_ne!(page_one_id, page_two_id); + assert!([first.as_str(), second.as_str()].contains(&page_one_id)); + assert!([first.as_str(), second.as_str()].contains(&page_two_id)); + + client + .shutdown() + .await + .expect("MCP client should shut down"); + harness.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_hides_archived_runs_by_default() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let active_id = unique_run_id(); + let archived_id = unique_run_id(); + let active = remote_run_summary_json( + &active_id, + "Simple", + "simple", + "Active run", + &serde_json::json!({ "kind": "succeeded", "reason": "completed" }), + "2026-04-05T12:00:00Z", + ); + let mut archived = remote_run_summary_json( + &archived_id, + "Simple", + "simple", + "Archived run", + &serde_json::json!({ "kind": "succeeded", "reason": "completed" }), + "2026-04-05T12:01:00Z", + ); + archived["lifecycle"]["archived"] = serde_json::json!(true); + archived["lifecycle"]["archived_at"] = serde_json::json!("2026-04-05T12:02:00Z"); + let active_resolve = mock_resolved_run_json(&server, &active_id, active, None); + let archived_resolve = mock_resolved_run_json(&server, &archived_id, archived, None); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let result = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [active_id, archived_id], "first": 10 }), + ) + .await; + + assert_eq!(result["runs"].as_array().unwrap().len(), 1); + assert!( + result["runs"] + .as_array() + .unwrap() + .iter() + .all(|run| run["archived"] == false) + ); + active_resolve.assert(); + archived_resolve.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_refreshes_expired_oauth_token() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_oauth_auth( + &context.home_dir, + &target, + "expired-access", + "refresh-octocat", + ); + let run_id = unique_run_id(); + let expired_access = server.mock(|when, then| { + when.method(GET) + .path("/api/v1/runs/resolve") + .query_param("selector", run_id.clone()) + .header("authorization", "Bearer expired-access"); + then.status(401) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "errors": [{ + "detail": "access token expired", + "code": "access_token_expired" + }] + })); + }); + let refresh = server.mock(|when, then| { + when.method(POST) + .path("/auth/cli/refresh") + .header("authorization", "Bearer refresh-octocat"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "access_token": "fresh-access", + "access_token_expires_at": (Utc::now() + ChronoDuration::minutes(10)).to_rfc3339(), + "refresh_token": "fresh-refresh", + "refresh_token_expires_at": (Utc::now() + ChronoDuration::days(30)).to_rfc3339(), + "subject": { + "idp_issuer": "https://github.com", + "idp_subject": "12345", + "login": "octocat", + "name": "The Octocat", + "email": "octocat@example.com" + } + })); + }); + let fresh_access = server.mock(|when, then| { + when.method(GET) + .path("/api/v1/runs/resolve") + .header("authorization", "Bearer fresh-access") + .query_param("selector", run_id.clone()); + then.status(200) + .header("Content-Type", "application/json") + .json_body(remote_run_summary_json( + &run_id, + "Simple", + "simple", + "OAuth refreshed", + &serde_json::json!({ "kind": "submitted" }), + "2026-04-05T12:00:00Z", + )); + }); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let result = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "first": 1 }), + ) + .await; + + assert_eq!(result["runs"][0]["run_id"], run_id); + expired_access.assert(); + refresh.assert(); + fresh_access.assert(); + let stored = AuthStore::new(context.home_dir.join(".fabro/auth.json")) + .get(&target) + .unwrap() + .unwrap(); + let AuthEntry::OAuth(stored) = stored else { + panic!("expected refreshed OAuth entry"); + }; + assert_eq!(stored.access_token, "fresh-access"); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_uses_fabro_auth_file_override() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + let auth_file = context.temp_dir.join("custom-auth.json"); + AuthStore::new(auth_file.clone()) + .put( + &target, + AuthEntry::DevToken(DevTokenEntry { + token: TEST_DEV_TOKEN.to_string(), + logged_in_at: Utc::now(), + }), + ) + .expect("custom auth store should be seeded"); + let run_id = unique_run_id(); + let authorization = format!("Bearer {TEST_DEV_TOKEN}"); + let resolve = mock_resolved_run_json( + &server, + &run_id, + remote_run_summary_json( + &run_id, + "Simple", + "simple", + "Custom auth file", + &serde_json::json!({ "kind": "submitted" }), + "2026-04-05T12:00:00Z", + ), + Some(&authorization), + ); + let mut fixture = mcp_stdio_fixture(&context, &["--server", &target_url]); + fixture.env.insert( + "FABRO_AUTH_FILE".to_string(), + auth_file.display().to_string(), + ); + let client = spawn_mcp_client_from_fixture(fixture).await; + + let result = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "first": 1 }), + ) + .await; + + assert_eq!(result["runs"][0]["run_id"], run_id); + resolve.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_orders_by_started_timestamp_before_created_timestamp() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let submitted_id = unique_run_id(); + let running_id = unique_run_id(); + let submitted = remote_run_summary_json( + &submitted_id, + "Simple", + "simple", + "Submitted later", + &serde_json::json!({ "kind": "submitted" }), + "2026-04-05T12:10:00Z", + ); + let mut running = remote_run_summary_json( + &running_id, + "Simple", + "simple", + "Started later", + &serde_json::json!({ "kind": "running" }), + "2026-04-05T12:00:00Z", + ); + running["timestamps"]["started_at"] = serde_json::json!("2026-04-05T12:20:00Z"); + let submitted_resolve = mock_resolved_run_json(&server, &submitted_id, submitted, None); + let running_resolve = mock_resolved_run_json(&server, &running_id, running, None); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let result = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [submitted_id, running_id], "first": 2 }), + ) + .await; + + assert_eq!(result["runs"][0]["run_id"], running_id); + assert_eq!(result["runs"][1]["run_id"], submitted_id); + submitted_resolve.assert(); + running_resolve.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_search_orders_submitted_runs_by_created_timestamp_not_run_id_timestamp() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let newer_created_id = run_id_with_timestamp("2026-04-05T12:00:00Z", 1); + let older_created_id = run_id_with_timestamp("2026-04-05T12:40:00Z", 1); + let mut newer_created = remote_run_summary_json( + &newer_created_id, + "Simple", + "simple", + "Created later", + &serde_json::json!({ "kind": "submitted" }), + "2026-04-05T12:30:00Z", + ); + newer_created["timestamps"]["started_at"] = serde_json::Value::Null; + let mut older_created = remote_run_summary_json( + &older_created_id, + "Simple", + "simple", + "Created earlier", + &serde_json::json!({ "kind": "submitted" }), + "2026-04-05T12:10:00Z", + ); + older_created["timestamps"]["started_at"] = serde_json::Value::Null; + let newer_resolve = mock_resolved_run_json(&server, &newer_created_id, newer_created, None); + let older_resolve = mock_resolved_run_json(&server, &older_created_id, older_created, None); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let result = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [newer_created_id, older_created_id], "first": 2 }), + ) + .await; + + assert_eq!(result["runs"][0]["run_id"], newer_created_id); + assert_eq!(result["runs"][1]["run_id"], older_created_id); + newer_resolve.assert(); + older_resolve.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_lifecycle_tools_manage_real_run() { + let context = test_context!(); + let harness = + RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let workflow = context.install_fixture("simple.fabro"); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let run_id = create_mcp_run(&client, workflow, true).await; + let cancel = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": run_id, "action": "cancel" }), + ) + .await; + + let gather = call_tool_json( + &client, + "fabro_run_gather", + serde_json::json!({ + "run_ids": [run_id], + "timeout_seconds": 20, + "poll_interval_seconds": 5 + }), + ) + .await; + let run_id = gather["runs"][0]["run_id"].as_str().unwrap().to_string(); + let get = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": run_id, "action": "get" }), + ) + .await; + let events = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ "run_id": run_id, "action": "list", "first": 5 }), + ) + .await; + let archive = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": run_id, "action": "archive" }), + ) + .await; + let archived_search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "archived": true }), + ) + .await; + let unarchive = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": run_id, "action": "unarchive" }), + ) + .await; + let search = call_tool_json( + &client, + "fabro_run_search", + serde_json::json!({ "run_ids": [run_id], "archived": false }), + ) + .await; + + fabro_json_snapshot!( + context, + serde_json::json!({ + "gather": normalize_gather(gather), + "cancel_action": cancel["action"], + "get_status": get["result"]["summary"]["status"], + "events_nonempty": events["events"].as_array().is_some_and(|events| !events.is_empty()), + "archive_action": archive["action"], + "archived_search_count": archived_search["runs"].as_array().unwrap().len(), + "archived_search_archived": archived_search["runs"][0]["archived"], + "unarchive_action": unarchive["action"], + "unarchived_search_count": search["runs"].as_array().unwrap().len(), + }), + @r#" + { + "gather": { + "runs": [ + { + "run_id": "[RUN_ID]", + "workflow_name": "Simple", + "workflow_slug": "simple", + "status": "failed", + "archived": false, + "created_at": "[TIMESTAMP]", + "started_at": null, + "completed_at": "[TIMESTAMP]", + "labels": { + "source": "mcp-test" + }, + "source_directory": "[SOURCE_DIRECTORY]", + "repo_origin_url": null, + "goal": "Run tests and report results" + } + ], + "timed_out": false, + "elapsed_seconds": "[ELAPSED]" + }, + "cancel_action": "cancel", + "get_status": "failed", + "events_nonempty": true, + "archive_action": "archive", + "archived_search_count": 1, + "archived_search_archived": true, + "unarchive_action": "unarchive", + "unarchived_search_count": 1 + } + "# + ); + + client + .shutdown() + .await + .expect("MCP client should shut down"); + harness.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_gather_rejects_too_many_runs() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + let run_ids = (0..51) + .map(|index| format!("run_{index}")) + .collect::>(); + + let error = call_tool_error_text( + &client, + "fabro_run_gather", + serde_json::json!({ "run_ids": run_ids }), + ) + .await; + + assert!(error.contains("run_ids"), "{error}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_gather_rejects_invalid_timeout_values_before_auth() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + + let timeout_error = call_tool_error_text( + &client, + "fabro_run_gather", + serde_json::json!({ + "run_ids": ["run_123"], + "timeout_seconds": 601, + "poll_interval_seconds": 5 + }), + ) + .await; + let poll_error = call_tool_error_text( + &client, + "fabro_run_gather", + serde_json::json!({ + "run_ids": ["run_123"], + "timeout_seconds": 300, + "poll_interval_seconds": 4 + }), + ) + .await; + + assert!(timeout_error.contains("timeout_seconds"), "{timeout_error}"); + assert!( + !timeout_error.contains("fabro auth login"), + "{timeout_error}" + ); + assert!(poll_error.contains("poll_interval_seconds"), "{poll_error}"); + assert!(!poll_error.contains("fabro auth login"), "{poll_error}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_gather_returns_timeout_result() { + let context = test_context!(); + let harness = + RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let workflow = context.install_fixture("simple.fabro"); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let run_id = create_mcp_run(&client, workflow, false).await; + + let start = std::time::Instant::now(); + let gather = call_tool_json( + &client, + "fabro_run_gather", + serde_json::json!({ + "run_ids": [run_id], + "timeout_seconds": 1, + "poll_interval_seconds": 5 + }), + ) + .await; + + assert_eq!(gather["timed_out"], true); + assert!(start.elapsed() < std::time::Duration::from_secs(4)); + assert_eq!(gather["runs"][0]["status"], "submitted"); + + client + .shutdown() + .await + .expect("MCP client should shut down"); + harness.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_interact_error_does_not_stop_server() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + + let error = call_tool_error_text( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": "run_123", "action": "message" }), + ) + .await; + + assert!(error.contains("message"), "{error}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_interact_actions_resolve_selector_and_call_expected_endpoints() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let selector = "nightly"; + let resolve = mock_resolved_run(&server, selector, &run_id); + let retrieve = server.mock(|when, then| { + when.method(GET).path(format!("/api/v1/runs/{run_id}")); + then.status(200) + .header("Content-Type", "application/json") + .json_body(remote_run_summary_json( + &run_id, + "Simple", + "simple", + "Run tests", + &serde_json::json!({ "kind": "running" }), + "2026-04-05T12:00:00Z", + )); + }); + let projection = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/state")); + then.status(200) + .header("Content-Type", "application/json") + .json_body(run_projection_json( + &run_id, + &serde_json::json!({ "kind": "running" }), + )); + }); + let start = server.mock(|when, then| { + when.method(POST) + .path(format!("/api/v1/runs/{run_id}/start")) + .json_body(serde_json::json!({ "resume": false })); + then.status(200) + .header("Content-Type", "application/json") + .json_body(remote_run_summary_json( + &run_id, + "Simple", + "simple", + "Run tests", + &serde_json::json!({ "kind": "running" }), + "2026-04-05T12:00:00Z", + )); + }); + let message = server.mock(|when, then| { + when.method(POST) + .path(format!("/api/v1/runs/{run_id}/steer")) + .json_body(serde_json::json!({ "text": "continue", "interrupt": true })); + then.status(202); + }); + let cancel = server.mock(|when, then| { + when.method(POST) + .path(format!("/api/v1/runs/{run_id}/cancel")); + then.status(200) + .header("Content-Type", "application/json") + .json_body(remote_run_summary_json( + &run_id, + "Simple", + "simple", + "Run tests", + &serde_json::json!({ "kind": "running" }), + "2026-04-05T12:00:00Z", + )); + }); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let get = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": selector, "action": "get" }), + ) + .await; + let start_result = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": selector, "action": "start" }), + ) + .await; + let message_result = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ + "run_id": selector, + "action": "message", + "message": "continue", + "interrupt": true + }), + ) + .await; + let cancel_result = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": selector, "action": "cancel" }), + ) + .await; + + assert_eq!(get["result"]["summary"]["run_id"], run_id); + assert_eq!(start_result["result"]["summary"]["run_id"], run_id); + assert_eq!(message_result["result"]["message"], "continue"); + assert_eq!(message_result["result"]["interrupt"], true); + assert_eq!(cancel_result["result"]["summary"]["run_id"], run_id); + resolve.assert_calls(4); + retrieve.assert_calls(1); + projection.assert(); + start.assert(); + message.assert(); + cancel.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_create_validation_errors_happen_before_auth_or_network() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + let too_many = (0..51) + .map(|index| serde_json::json!({ "workflow": format!("wf-{index}.fabro") })) + .collect::>(); + + let empty = call_tool_error_text( + &client, + "fabro_run_create", + serde_json::json!({ "runs": [] }), + ) + .await; + let many = call_tool_error_text( + &client, + "fabro_run_create", + serde_json::json!({ "runs": too_many }), + ) + .await; + let null = call_tool_error_text( + &client, + "fabro_run_create", + serde_json::json!({ + "runs": [{ + "workflow": "simple.fabro", + "inputs": { "decision": null } + }] + }), + ) + .await; + + assert!(empty.contains("runs"), "{empty}"); + assert!(many.contains("runs"), "{many}"); + assert!(null.contains("decision"), "{null}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_interact_answer_validation_happens_before_auth_or_network() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + + let error = call_tool_error_text( + &client, + "fabro_run_interact", + serde_json::json!({ + "run_id": "nightly", + "action": "answer", + "question_id": "q-1", + "answer": { "value": "yes" } + }), + ) + .await; + + assert!(error.contains("option, options, text"), "{error}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_interact_questions_and_answers_use_api_wire_contract() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let selector = "nightly"; + let resolve = mock_resolved_run(&server, selector, &run_id); + let questions = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/questions")) + .query_param("page[limit]", "100") + .query_param("page[offset]", "0"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": [{ + "id": "q-1", + "text": "Proceed?", + "stage": "gate", + "question_type": "yes_no", + "options": [], + "allow_freeform": false, + "timeout_seconds": null, + "context_display": null + }], + "meta": { "has_more": false } + })); + }); + let expected_answers = [ + ( + serde_json::json!(true), + serde_json::json!({ "kind": "yes" }), + ), + ( + serde_json::json!(false), + serde_json::json!({ "kind": "no" }), + ), + ( + serde_json::json!("Looks good"), + serde_json::json!({ "kind": "text", "text": "Looks good" }), + ), + ( + serde_json::json!({ "option": "approve" }), + serde_json::json!({ "kind": "selected", "option_key": "approve" }), + ), + ( + serde_json::json!({ "options": ["approve", "notify"] }), + serde_json::json!({ "kind": "multi_selected", "option_keys": ["approve", "notify"] }), + ), + ( + serde_json::json!({ "text": "Freeform" }), + serde_json::json!({ "kind": "text", "text": "Freeform" }), + ), + ]; + let answer_mocks = expected_answers + .iter() + .map(|(_, expected_body)| { + server.mock(|when, then| { + when.method(POST) + .path(format!("/api/v1/runs/{run_id}/questions/q-1/answer")) + .json_body(expected_body.clone()); + then.status(204); + }) + }) + .collect::>(); + + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + let question_result = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ "run_id": selector, "action": "get_questions" }), + ) + .await; + assert_eq!(question_result["result"]["questions"][0]["id"], "q-1"); + + for (answer, _) in expected_answers { + let result = call_tool_json( + &client, + "fabro_run_interact", + serde_json::json!({ + "run_id": selector, + "action": "answer", + "question_id": "q-1", + "answer": answer + }), + ) + .await; + assert_eq!(result["result"]["submitted"], true); + } + + resolve.assert_calls(7); + questions.assert(); + for answer in answer_mocks { + answer.assert(); + } + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_events_filters_find_matches_beyond_first_page() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let resolve = mock_resolved_run(&server, "nightly", &run_id); + let events = (1..=60) + .map(|sequence| { + let event_name = if sequence == 60 { + "stage.started" + } else { + "run.started" + }; + let properties = if sequence == 60 { + serde_json::json!({ + "index": 1, + "handler_type": "prompt", + "attempt": 1, + "max_attempts": 1 + }) + } else { + serde_json::json!({ + "name": "Simple", + "goal": format!("ordinary event {sequence}") + }) + }; + serde_json::json!({ + "seq": sequence, + "id": format!("evt-{sequence}"), + "ts": "2026-04-05T12:00:00Z", + "run_id": run_id, + "event": event_name, + "properties": properties, + "actor": null + }) + }) + .collect::>(); + let first_event = events[0].clone(); + let _limited_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param("limit", "1"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": [first_event], + "meta": { "has_more": true } + })); + }); + let list_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param_missing("limit"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": events, + "meta": { "has_more": false } + })); + }); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let details = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "details", + "event_ids": ["evt-60"], + "first": 1 + }), + ) + .await; + let filtered = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "search", + "categories": ["stage"], + "query": "prompt", + "first": 1, + "max_content_length": 32 + }), + ) + .await; + + assert_eq!(details["events"][0]["event_id"], "evt-60"); + assert_eq!(filtered["events"][0]["event_id"], "evt-60"); + assert_eq!(filtered["events"][0]["truncated"], true); + resolve.assert_calls(2); + list_events.assert_calls(2); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_events_requires_action_specific_inputs_before_auth() { + let context = test_context!(); + let client = spawn_mcp_client(&context, &["--server", "http://127.0.0.1:9"]).await; + + let details_error = call_tool_error_text( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "run_123", + "action": "details" + }), + ) + .await; + let search_error = call_tool_error_text( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "run_123", + "action": "search" + }), + ) + .await; + + assert!(details_error.contains("event_ids"), "{details_error}"); + assert!( + !details_error.contains("fabro auth login"), + "{details_error}" + ); + assert!(search_error.contains("query"), "{search_error}"); + assert!(!search_error.contains("fabro auth login"), "{search_error}"); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_events_desc_after_offset_and_limit_page_over_requested_order() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let resolve = mock_resolved_run(&server, "nightly", &run_id); + let events = (1..=5) + .map(|sequence| { + serde_json::json!({ + "seq": sequence, + "id": format!("evt-{sequence}"), + "ts": format!("2026-04-05T12:00:0{sequence}Z"), + "run_id": run_id, + "event": "run.started", + "properties": { "name": format!("event {sequence}") }, + "actor": null + }) + }) + .collect::>(); + let first_event = events[0].clone(); + let limited_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param("limit", "1"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": [first_event], + "meta": { "has_more": true } + })); + }); + let full_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param_missing("limit") + .query_param_missing("since_seq"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": events, + "meta": { "has_more": false } + })); + }); + let after_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param("since_seq", "2") + .query_param("limit", "3"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": [ + { + "seq": 2, + "id": "evt-2", + "ts": "2026-04-05T12:00:02Z", + "run_id": run_id, + "event": "run.started", + "properties": { "name": "event 2" }, + "actor": null + }, + { + "seq": 3, + "id": "evt-3", + "ts": "2026-04-05T12:00:03Z", + "run_id": run_id, + "event": "run.started", + "properties": { "name": "event 3" }, + "actor": null + }, + { + "seq": 4, + "id": "evt-4", + "ts": "2026-04-05T12:00:04Z", + "run_id": run_id, + "event": "run.started", + "properties": { "name": "event 4" }, + "actor": null + } + ], + "meta": { "has_more": true } + })); + }); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let desc = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "list", + "direction": "desc", + "first": 1 + }), + ) + .await; + let paged = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "list", + "after": 2, + "offset": 1, + "limit": 2 + }), + ) + .await; + + assert_eq!(desc["events"][0]["event_id"], "evt-5"); + assert_eq!(desc["next_cursor"], 5); + assert_eq!(paged["events"][0]["event_id"], "evt-3"); + assert_eq!(paged["events"][1]["event_id"], "evt-4"); + assert_eq!(paged["next_cursor"], 5); + resolve.assert_calls(2); + limited_events.assert_calls(0); + full_events.assert(); + after_events.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_events_desc_cursor_continues_to_older_events() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let resolve = mock_resolved_run(&server, "nightly", &run_id); + let events = (1..=5) + .map(|sequence| { + serde_json::json!({ + "seq": sequence, + "id": format!("evt-{sequence}"), + "ts": format!("2026-04-05T12:00:0{sequence}Z"), + "run_id": run_id, + "event": "run.started", + "properties": { "name": format!("event {sequence}") }, + "actor": null + }) + }) + .collect::>(); + let full_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param_missing("limit") + .query_param_missing("since_seq"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": events, + "meta": { "has_more": false } + })); + }); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let first_page = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "list", + "direction": "desc", + "first": 2 + }), + ) + .await; + let second_page = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "list", + "direction": "desc", + "after": first_page["next_cursor"], + "first": 2 + }), + ) + .await; + + assert_eq!(first_page["events"][0]["event_id"], "evt-5"); + assert_eq!(first_page["events"][1]["event_id"], "evt-4"); + assert_eq!(first_page["next_cursor"], 4); + assert_eq!(second_page["events"][0]["event_id"], "evt-3"); + assert_eq!(second_page["events"][1]["event_id"], "evt-2"); + assert_eq!(second_page["next_cursor"], 2); + resolve.assert_calls(2); + full_events.assert_calls(2); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_events_offset_beyond_fetch_cap_reaches_later_pages() { + let context = test_context!(); + let server = MockServer::start(); + let target_url = format!("{}/api/v1", server.base_url()); + let target: fabro_client::ServerTarget = target_url.parse().unwrap(); + seed_dev_token_auth(&context.home_dir, &target, TEST_DEV_TOKEN); + let run_id = unique_run_id(); + let resolve = mock_resolved_run(&server, "nightly", &run_id); + let events = (1..=300) + .map(|sequence| { + serde_json::json!({ + "seq": sequence, + "id": format!("evt-{sequence}"), + "ts": "2026-04-05T12:00:00Z", + "run_id": run_id, + "event": "run.started", + "properties": { "name": format!("event {sequence}") }, + "actor": null + }) + }) + .collect::>(); + let first_251_events = events.iter().take(251).cloned().collect::>(); + let bounded_events = server.mock(|when, then| { + when.method(GET) + .path(format!("/api/v1/runs/{run_id}/events")) + .query_param("limit", "251"); + then.status(200) + .header("Content-Type", "application/json") + .json_body(serde_json::json!({ + "data": first_251_events, + "meta": { "has_more": true } + })); + }); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let paged = call_tool_json( + &client, + "fabro_run_events", + serde_json::json!({ + "run_id": "nightly", + "action": "list", + "offset": 250, + "first": 1 + }), + ) + .await; + + assert_eq!(paged["events"][0]["event_id"], "evt-251"); + assert_eq!(paged["next_cursor"], 252); + resolve.assert(); + bounded_events.assert(); + client + .shutdown() + .await + .expect("MCP client should shut down"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn mcp_tool_auth_error_mentions_login() { + let context = test_context!(); + let harness = + RealAuthHarness::start_with_dev_token(fabro_test::GitHubAppState::default()).await; + let target_url = harness.api_target(); + let client = spawn_mcp_client(&context, &["--server", &target_url]).await; + + let error = call_tool_error_text( + &client, + "fabro_run_search", + serde_json::json!({ "first": 1 }), + ) + .await; + + assert!( + error.contains("Run `fabro auth login` to authenticate."), + "{error}" + ); + assert_eq!(client.list_tools().await.unwrap().len(), 5); + + client + .shutdown() + .await + .expect("MCP client should shut down"); + harness.shutdown().await; +} + +fn expected_claude_desktop_config_path(home_dir: &Path) -> PathBuf { + #[cfg(target_os = "macos")] + { + home_dir + .join("Library") + .join("Application Support") + .join("Claude") + .join("claude_desktop_config.json") + } + #[cfg(target_os = "linux")] + { + home_dir + .join(".config") + .join("Claude") + .join("claude_desktop_config.json") + } + #[cfg(target_os = "windows")] + { + home_dir + .join("AppData") + .join("Roaming") + .join("Claude") + .join("claude_desktop_config.json") + } +} + +struct McpStdioFixture { + command: Vec, + env: HashMap, + current_dir: PathBuf, +} + +fn mcp_stdio_fixture(context: &fabro_test::TestContext, extra_args: &[&str]) -> McpStdioFixture { + let mut command = vec![ + env!("CARGO_BIN_EXE_fabro").to_string(), + "mcp".to_string(), + "start".to_string(), + ]; + command.extend(extra_args.iter().map(|arg| (*arg).to_string())); + + let mut env = fabro_test::isolated_env(&context.home_dir); + env.insert( + "FABRO_HOME".to_string(), + context.home_dir.join(".fabro").display().to_string(), + ); + + McpStdioFixture { + command, + env, + current_dir: context.temp_dir.clone(), + } +} + +fn write_mcp_server_settings( + context: &mut fabro_test::TestContext, + storage_dir: &Path, + socket_path: Option<&Path>, +) { + context.manage_storage_dir(storage_dir); + let cli_target = socket_path.map_or_else(String::new, |path| { + format!( + r#" +[cli.target] +type = "unix" +path = "{}" +"#, + path.display() + ) + }); + context.write_home( + ".fabro/settings.toml", + format!( + r#"_version = 1 + +[server.storage] +root = "{}" + +[server.auth] +methods = ["dev-token"] +{cli_target}"#, + storage_dir.display() + ), + ); +} + +async fn spawn_mcp_client(context: &fabro_test::TestContext, extra_args: &[&str]) -> McpClient { + let fixture = mcp_stdio_fixture(context, extra_args); + spawn_mcp_client_from_fixture(fixture).await +} + +async fn spawn_mcp_client_from_fixture(fixture: McpStdioFixture) -> McpClient { + let config = McpServerSettings { + name: "fabro-under-test".to_string(), + transport: McpTransport::Stdio { + command: fixture.command, + env: fixture.env, + }, + current_dir: Some(fixture.current_dir), + clear_env: true, + startup_timeout_secs: 10, + tool_timeout_secs: 30, + }; + let client = McpClient::new(&config).expect("MCP client should build"); + client + .initialize(config.startup_timeout()) + .await + .expect("MCP server should initialize"); + client +} + +async fn call_tool_json( + client: &McpClient, + name: &str, + arguments: serde_json::Value, +) -> serde_json::Value { + let result = client + .call_tool(name, arguments, std::time::Duration::from_secs(30)) + .await + .expect("tool call should complete"); + assert_ne!( + result.is_error, + Some(true), + "tool returned error: {result:?}" + ); + let text = result + .content + .first() + .and_then(|content| serde_json::to_value(content).ok()) + .and_then(|content| content["text"].as_str().map(ToOwned::to_owned)) + .expect("tool result should include text fallback"); + assert!(!text.starts_with('{') && !text.starts_with('[')); + result + .structured_content + .expect("tool result should include structured content") +} + +async fn call_tool_error_text( + client: &McpClient, + name: &str, + arguments: serde_json::Value, +) -> String { + let result = client + .call_tool(name, arguments, std::time::Duration::from_secs(30)) + .await + .expect("tool call should complete"); + assert_eq!(result.is_error, Some(true), "tool should return error"); + result + .content + .first() + .and_then(|content| serde_json::to_value(content).ok()) + .and_then(|content| content["text"].as_str().map(ToOwned::to_owned)) + .expect("tool error should include text") +} + +async fn create_mcp_run(client: &McpClient, workflow: PathBuf, start: bool) -> String { + let create = call_tool_json( + client, + "fabro_run_create", + serde_json::json!({ + "runs": [{ + "workflow": workflow, + "dry_run": true, + "auto_approve": true, + "labels": { "source": "mcp-test" }, + "start": start + }] + }), + ) + .await; + create["runs"][0]["run_id"] + .as_str() + .expect("create result should include run id") + .to_string() +} + +fn seed_oauth_auth( + home_dir: &Path, + target: &fabro_client::ServerTarget, + access_token: &str, + refresh_token: &str, +) { + let now = Utc::now(); + AuthStore::new(home_dir.join(".fabro/auth.json")) + .put( + target, + AuthEntry::OAuth(OAuthEntry { + access_token: access_token.to_string(), + access_token_expires_at: now - ChronoDuration::minutes(1), + refresh_token: refresh_token.to_string(), + refresh_token_expires_at: now + ChronoDuration::days(30), + subject: StoredSubject { + idp_issuer: "https://github.com".to_string(), + idp_subject: "12345".to_string(), + login: "octocat".to_string(), + name: "The Octocat".to_string(), + email: "octocat@example.com".to_string(), + }, + logged_in_at: now, + }), + ) + .unwrap_or_else(|err| panic!("failed to seed OAuth auth: {err}")); +} + +fn normalize_run_search(mut value: serde_json::Value) -> serde_json::Value { + if let Some(runs) = value["runs"].as_array_mut() { + for run in runs { + run["run_id"] = serde_json::json!("[RUN_ID]"); + run["created_at"] = serde_json::json!("[TIMESTAMP]"); + if run["started_at"].is_string() { + run["started_at"] = serde_json::json!("[TIMESTAMP]"); + } + if run["completed_at"].is_string() { + run["completed_at"] = serde_json::json!("[TIMESTAMP]"); + } + if run["source_directory"].is_string() { + run["source_directory"] = serde_json::json!("[SOURCE_DIRECTORY]"); + } + if run["repo_origin_url"].is_string() { + run["repo_origin_url"] = serde_json::json!("[REPO_ORIGIN_URL]"); + } + } + } + value +} + +fn normalize_gather(mut value: serde_json::Value) -> serde_json::Value { + value["elapsed_seconds"] = serde_json::json!("[ELAPSED]"); + if let Some(runs) = value["runs"].as_array_mut() { + for run in runs { + run["run_id"] = serde_json::json!("[RUN_ID]"); + run["created_at"] = serde_json::json!("[TIMESTAMP]"); + if run["started_at"].is_string() { + run["started_at"] = serde_json::json!("[TIMESTAMP]"); + } + if run["completed_at"].is_string() { + run["completed_at"] = serde_json::json!("[TIMESTAMP]"); + } + if run["source_directory"].is_string() { + run["source_directory"] = serde_json::json!("[SOURCE_DIRECTORY]"); + } + if run["repo_origin_url"].is_string() { + run["repo_origin_url"] = serde_json::json!("[REPO_ORIGIN_URL]"); + } + } + } + value +} + +fn run_id_with_timestamp(timestamp: &str, sequence: u128) -> String { + let timestamp = DateTime::parse_from_rfc3339(timestamp) + .expect("test timestamp should parse") + .with_timezone(&Utc); + RunId::with_timestamp(timestamp, sequence).to_string() +} + +fn mock_resolved_run_json<'a>( + server: &'a MockServer, + selector: &str, + body: serde_json::Value, + authorization: Option<&str>, +) -> httpmock::Mock<'a> { + server.mock(|when, then| { + let when = when + .method(GET) + .path("/api/v1/runs/resolve") + .query_param("selector", selector); + if let Some(authorization) = authorization { + when.header("authorization", authorization); + } + then.status(200) + .header("Content-Type", "application/json") + .json_body(body); + }) +} diff --git a/lib/crates/fabro-cli/tests/it/cmd/mod.rs b/lib/crates/fabro-cli/tests/it/cmd/mod.rs index 58a98b5e3..0dd5da69b 100644 --- a/lib/crates/fabro-cli/tests/it/cmd/mod.rs +++ b/lib/crates/fabro-cli/tests/it/cmd/mod.rs @@ -20,6 +20,7 @@ mod inspect; mod install; mod json_global; mod logs; +mod mcp; mod model; mod model_list; mod model_test; diff --git a/lib/crates/fabro-cli/tests/it/scenario/lifecycle.rs b/lib/crates/fabro-cli/tests/it/scenario/lifecycle.rs index 0eadec381..ee26e2fbf 100644 --- a/lib/crates/fabro-cli/tests/it/scenario/lifecycle.rs +++ b/lib/crates/fabro-cli/tests/it/scenario/lifecycle.rs @@ -26,14 +26,17 @@ fn local_run_lifecycle() { }; // 1. Run a workflow - cmd(&[ - "run", - "--auto-approve", - "--sandbox", - "local", - fixture("command_pipeline.fabro").to_str().unwrap(), - ]) - .success(); + context + .run_cmd() + .args([ + "--auto-approve", + "--sandbox", + "local", + fixture("command_pipeline.fabro").to_str().unwrap(), + ]) + .timeout(timeout_for("local")) + .assert() + .success(); // 2. ps -a --json — should list exactly one run let label = context.test_case_label(); diff --git a/lib/crates/fabro-cli/tests/it/workflow/acp.rs b/lib/crates/fabro-cli/tests/it/workflow/acp.rs new file mode 100644 index 000000000..dffd20e49 --- /dev/null +++ b/lib/crates/fabro-cli/tests/it/workflow/acp.rs @@ -0,0 +1,208 @@ +#![expect( + clippy::disallowed_methods, + reason = "integration test initializes an isolated git repository with the system git binary" +)] + +use fabro_acp::test_support::fake_acp_agent_script; +use fabro_auth::{AuthCredential, AuthDetails}; +use fabro_config::Storage; +use fabro_model::Provider; +use fabro_test::test_context; +use fabro_types::EventBody; +use fabro_vault::{SecretType, Vault}; + +use super::{find_run_dir, has_event, read_conclusion, run_events, run_state}; + +#[test] +fn acp_backend_workflow() { + let mut context = test_context!(); + context.write_home( + ".fabro/settings.toml", + "[server.auth]\nmethods = [\"dev-token\"]\n", + ); + context.isolated_server(); + seed_openai_vault(&context.storage_dir); + let fake_agent = write_fake_acp_agent(&context); + let acp_command = fake_acp_command_attr(&fake_agent); + let workflow = context.temp_dir.join("acp_backend.fabro"); + context.write_temp( + "acp_backend.fabro", + format!( + r#"digraph ACP {{ + graph [goal="Exercise ACP backend"] + start [shape=Mdiamond] + work [type="agent", backend="acp", provider="openai", model="fake-acp", prompt="write hello.txt", acp_command={acp_command}] + exit [shape=Msquare] + start -> work + work -> exit +}}"# + ), + ); + init_git_repo(&context.temp_dir); + + context + .run_cmd() + .args(["--auto-approve", "--sandbox", "local"]) + .arg(&workflow) + .assert() + .success(); + + let run_dir = find_run_dir(&context); + let conclusion = read_conclusion(&run_dir); + assert_eq!(conclusion["status"].as_str(), Some("succeeded")); + + let events = run_events(&run_dir); + assert!(has_event(&run_dir, "agent.acp.started")); + assert!(has_event(&run_dir, "agent.acp.completed")); + let completed = events + .iter() + .find_map(|event| match &event.event.body { + EventBody::StageCompleted(props) if event.event.node_id.as_deref() == Some("work") => { + Some(props) + } + _ => None, + }) + .expect("work stage should complete"); + assert_eq!(completed.response.as_deref(), Some("hello from acp")); + assert!( + completed + .files_touched + .iter() + .any(|file| file == "hello.txt"), + "files_touched should include hello.txt: {:?}", + completed.files_touched + ); + + let state = serde_json::to_value(run_state(&run_dir)).expect("run state should serialize"); + let stages = state["stages"] + .as_object() + .expect("run state should contain stages"); + assert!( + stages.values().any(|stage| { + stage["provider_used"]["mode"] == "acp" + && stage["provider_used"]["provider"] == "openai" + }), + "run projection should include ACP provider metadata: {stages:?}" + ); +} + +#[test] +fn acp_prompt_workflow_uses_acp_backend() { + let mut context = test_context!(); + context.write_home( + ".fabro/settings.toml", + "[server.auth]\nmethods = [\"dev-token\"]\n", + ); + context.isolated_server(); + seed_openai_vault(&context.storage_dir); + let fake_agent = write_fake_acp_agent(&context); + let acp_command = fake_acp_command_attr(&fake_agent); + let workflow = context.temp_dir.join("acp_prompt_backend.fabro"); + context.write_temp( + "acp_prompt_backend.fabro", + format!( + r#"digraph ACP {{ + graph [goal="Exercise ACP prompt backend"] + start [shape=Mdiamond] + prompt [type="prompt", backend="acp", provider="openai", model="fake-acp", project_memory=false, prompt="write hello.txt", acp_command={acp_command}] + exit [shape=Msquare] + start -> prompt + prompt -> exit +}}"# + ), + ); + init_git_repo(&context.temp_dir); + + context + .run_cmd() + .args(["--auto-approve", "--sandbox", "local"]) + .arg(&workflow) + .assert() + .success(); + + let run_dir = find_run_dir(&context); + let conclusion = read_conclusion(&run_dir); + assert_eq!(conclusion["status"].as_str(), Some("succeeded")); + + let events = run_events(&run_dir); + assert!(has_event(&run_dir, "agent.acp.started")); + assert!(has_event(&run_dir, "agent.acp.completed")); + assert!( + !has_event(&run_dir, "agent.session.activated"), + "ACP prompt should not activate an API-mode agent session" + ); + let completed = events + .iter() + .find_map(|event| match &event.event.body { + EventBody::StageCompleted(props) + if event.event.node_id.as_deref() == Some("prompt") => + { + Some(props) + } + _ => None, + }) + .expect("prompt stage should complete"); + assert_eq!(completed.response.as_deref(), Some("hello from acp")); + + let state = serde_json::to_value(run_state(&run_dir)).expect("run state should serialize"); + let stages = state["stages"] + .as_object() + .expect("run state should contain stages"); + assert!( + stages.values().any(|stage| { + stage["provider_used"]["mode"] == "acp" + && stage["provider_used"]["provider"] == "openai" + }), + "run projection should include ACP provider metadata: {stages:?}" + ); +} + +fn seed_openai_vault(storage_dir: &std::path::Path) { + let mut vault = + Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load"); + vault + .set( + "openai", + &serde_json::to_string(&AuthCredential { + provider: Provider::OpenAi, + details: AuthDetails::ApiKey { + key: "test-openai-key".to_string(), + }, + }) + .expect("OpenAI test credential should serialize"), + SecretType::Credential, + None, + ) + .expect("OpenAI credential should store in test vault"); +} + +fn write_fake_acp_agent(context: &fabro_test::TestContext) -> std::path::PathBuf { + context.write_temp("fake_acp_agent.py", fake_acp_agent_script()); + context.temp_dir.join("fake_acp_agent.py") +} + +fn fake_acp_command_attr(script_path: &std::path::Path) -> String { + let command = serde_json::json!({ + "type": "stdio", + "name": "fake", + "command": "python3", + "args": [script_path.to_string_lossy()], + "env": [{"name": "ACP_MODE", "value": "write_file"}], + }) + .to_string(); + format!("{command:?}") +} + +fn init_git_repo(dir: &std::path::Path) { + let output = std::process::Command::new("git") + .args(["init", "-q"]) + .current_dir(dir) + .output() + .expect("git init should run"); + assert!( + output.status.success(), + "git init failed\nstdout:\n{}\nstderr:\n{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); +} diff --git a/lib/crates/fabro-cli/tests/it/workflow/command_agent_mixed.rs b/lib/crates/fabro-cli/tests/it/workflow/command_agent_mixed.rs index 1718e8428..447ef05b4 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/command_agent_mixed.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/command_agent_mixed.rs @@ -49,8 +49,8 @@ fn scenario_command_agent_mixed(sandbox: &str) { let export_dir = dump_export(&context, &run_id_for(&run_dir)); let stdout = - std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("stdout.log")) - .expect("verify stdout.log should exist"); + std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("output.log")) + .expect("verify output.log should exist"); assert!( stdout.contains("SCENARIO_FLAG_42"), "verify stdout should contain SCENARIO_FLAG_42, got: {stdout}" diff --git a/lib/crates/fabro-cli/tests/it/workflow/command_pipeline.rs b/lib/crates/fabro-cli/tests/it/workflow/command_pipeline.rs index ff6e583a9..2f0d4c94e 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/command_pipeline.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/command_pipeline.rs @@ -49,8 +49,8 @@ fn scenario_command_pipeline(sandbox: &str) { let export_dir = dump_export(&context, &run_id_for(&run_dir)); let stdout1 = - std::fs::read_to_string(stage_dump_dir(&export_dir, "step1@1").join("stdout.log")) - .expect("step1 stdout.log should exist"); + std::fs::read_to_string(stage_dump_dir(&export_dir, "step1@1").join("output.log")) + .expect("step1 output.log should exist"); assert!( stdout1.contains("hello-from-step1"), "step1 stdout should contain hello-from-step1, got: {stdout1}" diff --git a/lib/crates/fabro-cli/tests/it/workflow/full_stack.rs b/lib/crates/fabro-cli/tests/it/workflow/full_stack.rs index fef6e0578..3ba6c9caa 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/full_stack.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/full_stack.rs @@ -74,8 +74,8 @@ fn scenario_full_stack(sandbox: &str) { // Verify node stdout should contain PASS let export_dir = dump_export(&context, &run_id_for(&run_dir)); let stdout = - std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("stdout.log")) - .expect("verify stdout.log should exist"); + std::fs::read_to_string(stage_dump_dir(&export_dir, "verify@1").join("output.log")) + .expect("verify output.log should exist"); assert!( stdout.contains("PASS"), "verify stdout should contain PASS, got: {stdout}" diff --git a/lib/crates/fabro-cli/tests/it/workflow/hooks.rs b/lib/crates/fabro-cli/tests/it/workflow/hooks.rs index c2eecb762..5245d1a7f 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/hooks.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/hooks.rs @@ -11,9 +11,15 @@ use std::process::Output; -use fabro_test::{TestMode, TwinScenario, TwinScenarios, TwinToolCall, test_context, twin_openai}; +use fabro_auth::{AuthCredential, AuthDetails}; +use fabro_config::Storage; +use fabro_model::Provider; +use fabro_test::{ + TestMode, TwinOpenAi, TwinScenario, TwinScenarios, TwinToolCall, test_context, twin_openai, +}; +use fabro_vault::{SecretType, Vault}; -use super::{find_run_dir, read_conclusion}; +use super::read_conclusion; async fn run_success_output(mut cmd: assert_cmd::Command) -> Output { tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone()) @@ -51,6 +57,73 @@ fn stage_provider() -> &'static str { } } +fn toml_path(path: &std::path::Path) -> String { + path.display() + .to_string() + .replace('\\', "\\\\") + .replace('"', "\\\"") +} + +fn twin_server_storage_dir(context: &fabro_test::TestContext) -> std::path::PathBuf { + context.temp_dir.join("hook-server-storage") +} + +fn settings_with_hook(context: &fabro_test::TestContext, hook: &str) -> String { + if TestMode::from_env().is_twin() { + format!( + r#"[server.storage] +root = "{}" + +[server.auth] +methods = ["dev-token"] + +{hook}"#, + toml_path(&twin_server_storage_dir(context)), + ) + } else { + hook.to_string() + } +} + +fn write_hook_settings(context: &fabro_test::TestContext, hook: &str) { + let settings = settings_with_hook(context, hook); + if settings.trim().is_empty() { + return; + } + context.write_home(".fabro/settings.toml", settings); +} + +fn seed_openai_vault(storage_dir: &std::path::Path, base_url: &str, api_key: &str) { + let mut vault = + Vault::load(Storage::new(storage_dir).secrets_path()).expect("test vault should load"); + vault + .set( + "openai", + &serde_json::to_string(&AuthCredential { + provider: Provider::OpenAi, + details: AuthDetails::ApiKey { + key: api_key.to_string(), + }, + }) + .expect("OpenAI test credential should serialize"), + SecretType::Credential, + None, + ) + .expect("OpenAI credential should store in test vault"); + vault + .set("OPENAI_BASE_URL", base_url, SecretType::Environment, None) + .expect("OpenAI base URL should store in test vault"); +} + +fn configure_twin_server( + context: &mut fabro_test::TestContext, + twin: &TwinOpenAi, + namespace: &str, +) { + seed_openai_vault(&twin_server_storage_dir(context), &twin.base_url, namespace); + context.isolated_server(); +} + fn write_workflow(context: &fabro_test::TestContext, name: &str, dot: &str) -> std::path::PathBuf { context.write_temp(name, dot); context.temp_dir.join(name) @@ -69,25 +142,28 @@ fn configure_hook_env(cmd: &mut assert_cmd::Command, hook_model: &str) { cmd.arg("--model").arg(hook_model); } -fn conclusion_status(context: &fabro_test::TestContext) -> String { - let run_dir = find_run_dir(&context); - read_conclusion(&run_dir)["status"] - .as_str() - .expect("conclusion should include a string status") - .to_string() +async fn conclusion_status(context: &fabro_test::TestContext) -> String { + let run_dir = context.single_run_dir(); + tokio::task::spawn_blocking(move || { + read_conclusion(&run_dir)["status"] + .as_str() + .expect("conclusion should include a string status") + .to_string() + }) + .await + .expect("conclusion status task should complete") } #[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))] async fn hook_prompt_proceed_allows_run() { - let context = test_context!(); - context.write_home( - ".fabro/settings.toml", + let mut context = test_context!(); + write_hook_settings( + &context, &format!( r#" -[[hooks]] +[[run.hooks]] name = "prompt-proceed" event = "run_start" -type = "prompt" prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}." model = "{model}" "#, @@ -111,6 +187,7 @@ model = "{model}" .scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#)) .load(twin) .await; + configure_twin_server(&mut context, twin, &namespace); let mut cmd = context.run_cmd(); configure_hook_env(&mut cmd, stage_model()); twin.configure_command(&mut cmd, &namespace); @@ -123,20 +200,19 @@ model = "{model}" run_success_output(cmd).await; } - assert_eq!(conclusion_status(&context), "succeeded"); + assert_eq!(conclusion_status(&context).await, "succeeded"); } #[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))] async fn hook_prompt_block_prevents_run() { - let context = test_context!(); - context.write_home( - ".fabro/settings.toml", + let mut context = test_context!(); + write_hook_settings( + &context, &format!( r#" -[[hooks]] +[[run.hooks]] name = "prompt-block" event = "run_start" -type = "prompt" prompt = "Check: is 2+2 equal to 5? If the statement is true, respond {{\"ok\": true}}. If false, respond {{\"ok\": false, \"reason\": \"math check failed\"}}." model = "{model}" "#, @@ -163,6 +239,7 @@ model = "{model}" ) .load(twin) .await; + configure_twin_server(&mut context, twin, &namespace); let mut cmd = context.run_cmd(); configure_hook_env(&mut cmd, stage_model()); twin.configure_command(&mut cmd, &namespace); @@ -184,18 +261,18 @@ model = "{model}" #[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))] async fn hook_agent_proceed_allows_run() { - let context = test_context!(); - context.write_home( - ".fabro/settings.toml", + let mut context = test_context!(); + write_hook_settings( + &context, &format!( r#" -[[hooks]] +[[run.hooks]] name = "agent-proceed" event = "run_start" -type = "agent" prompt = "A workflow is starting. Always approve. Respond with {{\"ok\": true}}. Do not use any tools." model = "{model}" max_tool_rounds = 1 +agent = "enabled" "#, model = hook_model() ), @@ -217,6 +294,7 @@ max_tool_rounds = 1 .scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#)) .load(twin) .await; + configure_twin_server(&mut context, twin, &namespace); let mut cmd = context.run_cmd(); configure_hook_env(&mut cmd, stage_model()); twin.configure_command(&mut cmd, &namespace); @@ -229,25 +307,25 @@ max_tool_rounds = 1 run_success_output(cmd).await; } - assert_eq!(conclusion_status(&context), "succeeded"); + assert_eq!(conclusion_status(&context).await, "succeeded"); } #[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))] async fn hook_agent_with_tool_use() { - let context = test_context!(); + let mut context = test_context!(); let marker = context.temp_dir.join("hook_check.txt"); std::fs::write(&marker, "READY").unwrap(); - context.write_home( - ".fabro/settings.toml", + write_hook_settings( + &context, &format!( r#" -[[hooks]] +[[run.hooks]] name = "agent-tools" event = "run_start" -type = "agent" prompt = "Read the file at {path} using the read_file tool. If it contains 'READY', respond with {{\"ok\": true}}. Otherwise respond with {{\"ok\": false, \"reason\": \"not ready\"}}." model = "{model}" max_tool_rounds = 5 +agent = "enabled" "#, path = marker.display(), model = hook_model() @@ -274,6 +352,7 @@ max_tool_rounds = 5 .scenario(TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#)) .load(twin) .await; + configure_twin_server(&mut context, twin, &namespace); let mut cmd = context.run_cmd(); configure_hook_env(&mut cmd, stage_model()); twin.configure_command(&mut cmd, &namespace); @@ -286,12 +365,13 @@ max_tool_rounds = 5 run_success_output(cmd).await; } - assert_eq!(conclusion_status(&context), "succeeded"); + assert_eq!(conclusion_status(&context).await, "succeeded"); } #[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))] async fn arc_e2e_with_real_llm() { - let context = test_context!(); + let mut context = test_context!(); + write_hook_settings(&context, ""); let hello = context.temp_dir.join("hello.txt"); let workflow = write_workflow( &context, @@ -328,6 +408,7 @@ async fn arc_e2e_with_real_llm() { ) .load(twin) .await; + configure_twin_server(&mut context, twin, &namespace); let mut cmd = context.run_cmd(); configure_hook_env(&mut cmd, stage_model()); twin.configure_command(&mut cmd, &namespace); @@ -345,5 +426,5 @@ async fn arc_e2e_with_real_llm() { "Hello from LLM", "workflow should create the expected file" ); - assert_eq!(conclusion_status(&context), "succeeded"); + assert_eq!(conclusion_status(&context).await, "succeeded"); } diff --git a/lib/crates/fabro-cli/tests/it/workflow/mod.rs b/lib/crates/fabro-cli/tests/it/workflow/mod.rs index 1d5c5a4e0..023d4f825 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/mod.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/mod.rs @@ -3,6 +3,7 @@ reason = "This test module prefers explicit type paths over extra imports." )] +mod acp; mod agent_linear; mod command_agent_mixed; mod command_pipeline; diff --git a/lib/crates/fabro-cli/tests/it/workflow/real_cli.rs b/lib/crates/fabro-cli/tests/it/workflow/real_cli.rs index efa322731..ee074c704 100644 --- a/lib/crates/fabro-cli/tests/it/workflow/real_cli.rs +++ b/lib/crates/fabro-cli/tests/it/workflow/real_cli.rs @@ -1,11 +1,10 @@ use std::sync::Arc; -use std::time::Duration; use fabro_graphviz::graph::{AttrValue, Node}; use fabro_llm::provider::Provider; use fabro_workflow::context::Context; use fabro_workflow::event::Emitter; -use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult}; +use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult, CodergenRunRequest}; use fabro_workflow::handler::llm::cli::AgentCliBackend; /// Run a real CLI tool via LocalSandbox and verify the full flow. @@ -14,8 +13,7 @@ async fn run_real_cli_test(provider: Provider, model: &str) { let env: Arc = Arc::new(fabro_agent::LocalSandbox::new( workspace.path().to_path_buf(), )); - let backend = AgentCliBackend::new_from_env(model.to_string(), provider) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env(model.to_string(), provider); let mut node = Node::new("real_cli_test"); node.attrs.insert( @@ -26,16 +24,16 @@ async fn run_real_cli_test(provider: Provider, model: &str) { let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( - &node, - "What is 2+2? Reply with just the number.", - &context, - None, - &emitter, - &env, - None, - tokio_util::sync::CancellationToken::new(), - ) + .run(CodergenRunRequest { + node: &node, + prompt: "What is 2+2? Reply with just the number.", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &env, + tool_hooks: None, + cancel_token: tokio_util::sync::CancellationToken::new(), + }) .await .unwrap_or_else(|_| panic!("CLI backend ({provider}/{model}) should succeed")); diff --git a/lib/crates/fabro-cli/tests/manifest_path_round_trip.rs b/lib/crates/fabro-cli/tests/manifest_path_round_trip.rs index 9cf4c50f8..d94f95221 100644 --- a/lib/crates/fabro-cli/tests/manifest_path_round_trip.rs +++ b/lib/crates/fabro-cli/tests/manifest_path_round_trip.rs @@ -5,7 +5,7 @@ use std::path::PathBuf; -use fabro_cli::{ManifestBuildInput, build_run_manifest}; +use fabro_manifest::{ManifestBuildInput, build_run_manifest}; use fabro_workflow::ManifestPath; #[test] diff --git a/lib/crates/fabro-client/src/client.rs b/lib/crates/fabro-client/src/client.rs index 97627466b..3654fd740 100644 --- a/lib/crates/fabro-client/src/client.rs +++ b/lib/crates/fabro-client/src/client.rs @@ -794,25 +794,27 @@ impl Client { Ok(bytes) } - pub async fn start_run(&self, run_id: &RunId, resume: bool) -> Result<()> { - self.send_api(|client| async move { - client - .start_run() - .id(run_id.to_string()) - .body(types::StartRunRequest { resume }) - .send() - .await - }) - .await?; - Ok(()) + pub async fn start_run(&self, run_id: &RunId, resume: bool) -> Result { + let response = self + .send_api(|client| async move { + client + .start_run() + .id(run_id.to_string()) + .body(types::StartRunRequest { resume }) + .send() + .await + }) + .await?; + convert_type(response.into_inner()) } - pub async fn cancel_run(&self, run_id: &RunId) -> Result<()> { - self.send_api( - |client| async move { client.cancel_run().id(run_id.to_string()).send().await }, - ) - .await?; - Ok(()) + pub async fn cancel_run(&self, run_id: &RunId) -> Result { + let response = self + .send_api( + |client| async move { client.cancel_run().id(run_id.to_string()).send().await }, + ) + .await?; + convert_type(response.into_inner()) } pub async fn interrupt_run(&self, run_id: &RunId) -> Result<()> { @@ -844,20 +846,22 @@ impl Client { Ok(()) } - pub async fn archive_run(&self, run_id: &RunId) -> Result<()> { - self.send_api( - |client| async move { client.archive_run().id(run_id.to_string()).send().await }, - ) - .await?; - Ok(()) + pub async fn archive_run(&self, run_id: &RunId) -> Result { + let response = self + .send_api( + |client| async move { client.archive_run().id(run_id.to_string()).send().await }, + ) + .await?; + convert_type(response.into_inner()) } - pub async fn unarchive_run(&self, run_id: &RunId) -> Result<()> { - self.send_api(|client| async move { - client.unarchive_run().id(run_id.to_string()).send().await - }) - .await?; - Ok(()) + pub async fn unarchive_run(&self, run_id: &RunId) -> Result { + let response = self + .send_api( + |client| async move { client.unarchive_run().id(run_id.to_string()).send().await }, + ) + .await?; + convert_type(response.into_inner()) } pub async fn rewind_run( @@ -1123,6 +1127,50 @@ impl Client { Ok(all_events) } + pub async fn list_run_events_until( + &self, + run_id: &RunId, + since_seq: Option, + max_events: usize, + ) -> Result> { + if max_events == 0 { + return Ok(Vec::new()); + } + + let mut next_since_seq = since_seq; + let mut all_events = Vec::new(); + while all_events.len() < max_events { + let remaining = max_events - all_events.len(); + let response = self + .send_api(|client| async move { + let mut request = client + .list_run_events() + .id(run_id.to_string()) + .limit(remaining.min(1000) as u64); + if let Some(seq) = next_since_seq.and_then(non_zero_u64_from_u32) { + request = request.since_seq(seq); + } + request.send().await + }) + .await?; + let parsed = response.into_inner(); + let page_events = parsed + .data + .into_iter() + .map(convert_type::<_, EventEnvelope>) + .collect::>>()?; + let next_page_since_seq = page_events.last().map(|event| event.seq.saturating_add(1)); + all_events.extend(page_events); + + if !parsed.meta.has_more || next_page_since_seq.is_none() { + break; + } + next_since_seq = next_page_since_seq; + } + + Ok(all_events) + } + pub async fn attach_run_events( &self, run_id: &RunId, diff --git a/lib/crates/fabro-config/src/resolve/run.rs b/lib/crates/fabro-config/src/resolve/run.rs index 4a3bc8815..2debd724d 100644 --- a/lib/crates/fabro-config/src/resolve/run.rs +++ b/lib/crates/fabro-config/src/resolve/run.rs @@ -354,6 +354,8 @@ pub(crate) fn resolve_mcp_entry(name: &str, entry: &McpEntryLayer) -> McpServerS McpServerSettings { name: name.to_string(), transport, + current_dir: None, + clear_env: false, startup_timeout_secs, tool_timeout_secs, } diff --git a/lib/crates/fabro-manifest/Cargo.toml b/lib/crates/fabro-manifest/Cargo.toml new file mode 100644 index 000000000..417dc9af0 --- /dev/null +++ b/lib/crates/fabro-manifest/Cargo.toml @@ -0,0 +1,29 @@ +[package] +name = "fabro-manifest" +edition.workspace = true +version.workspace = true +publish = false +license.workspace = true +description = "Fabro run manifest construction" + +[lib] +doctest = false + +[lints] +workspace = true + +[dependencies] +anyhow.workspace = true +fabro-api = { path = "../fabro-api" } +fabro-config = { path = "../fabro-config" } +fabro-github = { path = "../fabro-github" } +fabro-graphviz = { path = "../fabro-graphviz" } +fabro-template = { path = "../fabro-template" } +fabro-types = { path = "../fabro-types" } +fabro-workflow = { path = "../fabro-workflow" } +git2.workspace = true +toml.workspace = true + +[dev-dependencies] +tempfile = "3" +temp-env = "0.3" diff --git a/lib/crates/fabro-cli/src/manifest_builder.rs b/lib/crates/fabro-manifest/src/lib.rs similarity index 89% rename from lib/crates/fabro-cli/src/manifest_builder.rs rename to lib/crates/fabro-manifest/src/lib.rs index b2436db77..a176dbac6 100644 --- a/lib/crates/fabro-cli/src/manifest_builder.rs +++ b/lib/crates/fabro-manifest/src/lib.rs @@ -10,19 +10,21 @@ use anyhow::{Context, Result, anyhow}; use fabro_api::types; use fabro_config::project::{self, discover_project_config, resolve_workflow_path}; use fabro_config::run::{resolve_run_goal_from_layer, resolve_run_goal_from_namespace}; -use fabro_config::{CliLayer, DaytonaDockerfileLayer, RunLayer, WorkflowSettingsBuilder}; +use fabro_config::{ + CliLayer, DaytonaDockerfileLayer, DockerSandboxLayer, ReplaceMap, RunExecutionLayer, + RunGoalLayer, RunLayer, RunModelLayer, RunSandboxLayer, WorkflowSettingsBuilder, +}; use fabro_graphviz::graph::AttrValue; use fabro_graphviz::parser; use fabro_template::{TemplateContext, render as render_template}; -use fabro_types::settings::run::{ResolvedGoalSource, ResolvedRunGoal}; +use fabro_types::settings::interp::InterpString; +use fabro_types::settings::run::{ApprovalMode, ResolvedGoalSource, ResolvedRunGoal, RunMode}; use fabro_types::{DirtyStatus, GitContext, PreRunPushOutcome, RunId, WorkflowSettings}; use fabro_workflow::ManifestPath; use fabro_workflow::git::{ GitSyncStatus, branch_needs_push, head_sha, push_branch_noninteractive, sync_status, }; -use crate::args::{PreflightArgs, RunArgs}; - #[derive(Debug, Default)] pub struct ManifestBuildInput { pub workflow: PathBuf, @@ -43,6 +45,81 @@ pub struct BuiltManifest { pub target_path: PathBuf, } +#[derive(Debug, Default)] +pub struct RunOverrideInput<'a> { + pub goal: Option<&'a str>, + pub model: Option<&'a str>, + pub provider: Option<&'a str>, + pub sandbox: Option<&'a str>, + pub docker_image: Option<&'a str>, + pub preserve_sandbox: Option, + pub dry_run: Option, + pub auto_approve: Option, + pub labels: HashMap, +} + +#[must_use] +pub fn build_run_overrides(input: RunOverrideInput<'_>) -> RunLayer { + let goal = input + .goal + .map(|goal| RunGoalLayer::Inline(InterpString::parse(goal))); + let model = (input.model.is_some() || input.provider.is_some()).then(|| RunModelLayer { + provider: input.provider.map(InterpString::parse), + name: input.model.map(InterpString::parse), + fallbacks: Vec::new(), + controls: None, + }); + let sandbox = (input.sandbox.is_some() + || input.docker_image.is_some() + || input.preserve_sandbox.is_some()) + .then(|| RunSandboxLayer { + provider: input.sandbox.map(ToOwned::to_owned), + docker: input.docker_image.map(|image| DockerSandboxLayer { + image: Some(image.to_string()), + ..DockerSandboxLayer::default() + }), + preserve: input.preserve_sandbox, + ..RunSandboxLayer::default() + }); + let execution = + (input.dry_run.is_some() || input.auto_approve.is_some()).then(|| RunExecutionLayer { + mode: input.dry_run.map(|dry_run| { + if dry_run { + RunMode::DryRun + } else { + RunMode::Normal + } + }), + approval: input.auto_approve.map(|auto_approve| { + if auto_approve { + ApprovalMode::Auto + } else { + ApprovalMode::Prompt + } + }), + }); + + RunLayer { + goal, + metadata: ReplaceMap::from(input.labels), + model, + sandbox, + execution, + ..RunLayer::default() + } +} + +#[must_use] +pub fn build_sparse_run_overrides(input: RunOverrideInput<'_>) -> Option { + let run = build_run_overrides(input); + (run.goal.is_some() + || !run.metadata.is_empty() + || run.model.is_some() + || run.sandbox.is_some() + || run.execution.is_some()) + .then_some(run) +} + struct CollectContext<'a> { cwd: &'a Path, inputs: &'a HashMap, @@ -172,42 +249,6 @@ pub fn build_run_manifest(input: ManifestBuildInput) -> Result { }) } -pub(crate) fn run_manifest_args(args: &RunArgs) -> Option { - let payload = types::ManifestArgs { - auto_approve: args.auto_approve.then_some(true), - dry_run: args.dry_run.then_some(true), - label: args.label.clone(), - model: args.model.clone(), - preserve_sandbox: args.preserve_sandbox.then_some(true), - provider: args.provider.clone(), - sandbox: args - .sandbox - .map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()), - docker_image: None, - input: args.inputs.values.clone(), - verbose: args.verbose.then_some(true), - }; - (!manifest_args_is_empty(&payload)).then_some(payload) -} - -pub(crate) fn preflight_manifest_args(args: &PreflightArgs) -> Option { - let payload = types::ManifestArgs { - auto_approve: None, - dry_run: None, - label: Vec::new(), - model: args.model.clone(), - preserve_sandbox: None, - provider: args.provider.clone(), - sandbox: args - .sandbox - .map(|provider| fabro_sandbox::SandboxProvider::from(provider).to_string()), - docker_image: None, - input: args.inputs.values.clone(), - verbose: args.verbose.then_some(true), - }; - (!manifest_args_is_empty(&payload)).then_some(payload) -} - fn collect_workflow_entry( context: &mut CollectContext<'_>, workflow: &Path, @@ -650,7 +691,7 @@ fn manifest_path_from_absolute(path: &Path, cwd: &Path) -> Result .ok_or_else(|| anyhow!("Failed to compute manifest path for {}", path.display())) } -fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool { +pub fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool { args.auto_approve.is_none() && args.dry_run.is_none() && args.label.is_empty() @@ -667,6 +708,65 @@ fn manifest_args_is_empty(args: &types::ManifestArgs) -> bool { mod tests { use super::*; + #[test] + fn build_run_overrides_sets_common_cli_and_mcp_layers() { + let overrides = build_run_overrides(RunOverrideInput { + goal: Some("ship it"), + model: Some("gpt-5.4-mini"), + provider: Some("openai"), + sandbox: Some("local"), + docker_image: None, + preserve_sandbox: Some(true), + dry_run: Some(true), + auto_approve: Some(false), + labels: [("source".to_string(), "mcp".to_string())] + .into_iter() + .collect(), + }); + + let goal = overrides.goal.expect("goal override"); + assert!(matches!(goal, fabro_config::RunGoalLayer::Inline(_))); + assert_eq!( + overrides + .model + .as_ref() + .unwrap() + .name + .as_ref() + .unwrap() + .as_source(), + "gpt-5.4-mini" + ); + assert_eq!( + overrides + .model + .as_ref() + .unwrap() + .provider + .as_ref() + .unwrap() + .as_source(), + "openai" + ); + assert_eq!( + overrides.sandbox.as_ref().unwrap().provider.as_deref(), + Some("local") + ); + assert_eq!(overrides.sandbox.as_ref().unwrap().preserve, Some(true)); + assert_eq!( + overrides.execution.as_ref().unwrap().mode, + Some(RunMode::DryRun) + ); + assert_eq!( + overrides.execution.as_ref().unwrap().approval, + Some(ApprovalMode::Prompt) + ); + assert_eq!( + overrides.metadata.0.get("source").map(String::as_str), + Some("mcp") + ); + } + #[test] fn build_manifest_bundles_imports_prompts_and_children() { let temp = tempfile::tempdir().unwrap(); diff --git a/lib/crates/fabro-mcp-server/Cargo.toml b/lib/crates/fabro-mcp-server/Cargo.toml new file mode 100644 index 000000000..15f481431 --- /dev/null +++ b/lib/crates/fabro-mcp-server/Cargo.toml @@ -0,0 +1,31 @@ +[package] +name = "fabro-mcp-server" +edition.workspace = true +version.workspace = true +publish = false +license.workspace = true +description = "Fabro MCP stdio server" + +[lib] +doctest = false + +[lints] +workspace = true + +[dependencies] +anyhow.workspace = true +chrono = { workspace = true, features = ["serde"] } +fabro-api = { path = "../fabro-api" } +fabro-client = { path = "../fabro-client" } +fabro-manifest = { path = "../fabro-manifest" } +fabro-config = { path = "../fabro-config" } +fabro-server = { path = "../fabro-server" } +fabro-types = { path = "../fabro-types" } +fabro-util = { path = "../fabro-util" } +futures.workspace = true +rmcp = { workspace = true, features = ["server", "macros", "schemars", "transport-io"] } +schemars = "1.2.1" +serde.workspace = true +serde_json.workspace = true +tokio.workspace = true +toml.workspace = true diff --git a/lib/crates/fabro-mcp-server/src/config.rs b/lib/crates/fabro-mcp-server/src/config.rs new file mode 100644 index 000000000..9eb8a67c2 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/config.rs @@ -0,0 +1,146 @@ +#![expect( + clippy::disallowed_methods, + reason = "MCP client config setup intentionally performs small synchronous JSON file reads/writes from a CLI command." +)] + +use std::path::{Path, PathBuf}; + +use anyhow::{Context as _, Result, anyhow}; +use serde_json::map::Entry; +use serde_json::{Map, Value, json}; + +use crate::{McpAgent, McpConfigSettings, McpInitSettings}; + +const SERVER_NAME: &str = "fabro"; + +pub fn config_json(settings: &McpConfigSettings) -> Result { + serde_json::to_string_pretty(&generic_config(settings)) + .map(|json| format!("{json}\n")) + .context("failed to render Fabro MCP client config") +} + +pub fn init_agent(settings: &McpInitSettings) -> Result<()> { + let entry = server_entry(&settings.config); + for path in agent_config_paths(settings.agent, &settings.home_dir) { + merge_server_entry(&path, entry.clone())?; + } + Ok(()) +} + +fn generic_config(settings: &McpConfigSettings) -> Value { + json!({ + "mcpServers": { + SERVER_NAME: server_entry(settings) + } + }) +} + +fn server_entry(settings: &McpConfigSettings) -> Value { + json!({ + "command": "fabro", + "args": start_args(settings), + }) +} + +fn start_args(settings: &McpConfigSettings) -> Vec { + let mut args = vec!["mcp".to_string(), "start".to_string()]; + if let Some(server) = settings.server.as_ref() { + args.push("--server".to_string()); + args.push(server.clone()); + } + if let Some(storage_dir) = settings.storage_dir.as_deref() { + args.push("--storage-dir".to_string()); + args.push(storage_dir.display().to_string()); + } + args +} + +fn merge_server_entry(path: &Path, entry: Value) -> Result<()> { + if let Some(parent) = path.parent() { + std::fs::create_dir_all(parent) + .with_context(|| format!("failed to create {}", parent.display()))?; + } + + let mut root = match std::fs::read_to_string(path) { + Ok(contents) => serde_json::from_str::(&contents) + .with_context(|| format!("failed to parse MCP config {}", path.display()))?, + Err(err) if err.kind() == std::io::ErrorKind::NotFound => Value::Object(Map::new()), + Err(err) => return Err(err).with_context(|| format!("failed to read {}", path.display())), + }; + + let root_object = root + .as_object_mut() + .ok_or_else(|| anyhow!("MCP config {} must contain a JSON object", path.display()))?; + + let servers = match root_object.entry("mcpServers") { + Entry::Vacant(entry) => entry.insert(Value::Object(Map::new())), + Entry::Occupied(entry) => entry.into_mut(), + }; + let servers_object = servers.as_object_mut().ok_or_else(|| { + anyhow!( + "MCP config {} field mcpServers must contain a JSON object", + path.display() + ) + })?; + servers_object.insert(SERVER_NAME.to_string(), entry); + + let rendered = serde_json::to_string_pretty(&root) + .map(|json| format!("{json}\n")) + .with_context(|| format!("failed to render MCP config {}", path.display()))?; + std::fs::write(path, rendered).with_context(|| format!("failed to write {}", path.display())) +} + +fn agent_config_paths(agent: McpAgent, home_dir: &Path) -> Vec { + match agent { + McpAgent::Claude => vec![ + claude_desktop_config_path(home_dir), + claude_code_config_path(home_dir), + ], + McpAgent::Cursor => vec![home_dir.join(".cursor").join("mcp.json")], + McpAgent::Windsurf => vec![ + home_dir + .join(".codeium") + .join("windsurf") + .join("mcp_config.json"), + ], + } +} + +fn claude_code_config_path(home_dir: &Path) -> PathBuf { + home_dir.join(".claude.json") +} + +fn claude_desktop_config_path(home_dir: &Path) -> PathBuf { + #[cfg(target_os = "macos")] + { + home_dir + .join("Library") + .join("Application Support") + .join("Claude") + .join("claude_desktop_config.json") + } + + #[cfg(target_os = "linux")] + { + home_dir + .join(".config") + .join("Claude") + .join("claude_desktop_config.json") + } + + #[cfg(target_os = "windows")] + { + let app_data = std::env::var_os("APPDATA") + .map(PathBuf::from) + .unwrap_or_else(|| home_dir.join("AppData").join("Roaming")); + app_data.join("Claude").join("claude_desktop_config.json") + } + + #[cfg(not(any(target_os = "macos", target_os = "linux", target_os = "windows")))] + { + home_dir + .join(".config") + .join("Claude") + .join("claude_desktop_config.json") + } +} diff --git a/lib/crates/fabro-mcp-server/src/lib.rs b/lib/crates/fabro-mcp-server/src/lib.rs new file mode 100644 index 000000000..b8351bab5 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/lib.rs @@ -0,0 +1,55 @@ +mod config; +mod run_tools; +mod server; + +use std::future::Future; +use std::path::PathBuf; +use std::pin::Pin; +use std::sync::Arc; + +use anyhow::Result; +pub use config::{config_json, init_agent}; +use fabro_client::Client; +pub use server::start; + +pub type FabroClientFuture = Pin> + Send>>; + +pub type FabroClientFactory = Arc FabroClientFuture + Send + Sync>; + +#[derive(Clone)] +pub struct FabroMcpServerSettings { + pub client_factory: FabroClientFactory, + pub config_path: PathBuf, + pub cwd: PathBuf, +} + +impl std::fmt::Debug for FabroMcpServerSettings { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter + .debug_struct("FabroMcpServerSettings") + .field("client_factory", &"") + .field("config_path", &self.config_path) + .field("cwd", &self.cwd) + .finish() + } +} + +#[derive(Debug, Clone, Default)] +pub struct McpConfigSettings { + pub server: Option, + pub storage_dir: Option, +} + +#[derive(Debug, Clone)] +pub struct McpInitSettings { + pub agent: McpAgent, + pub config: McpConfigSettings, + pub home_dir: PathBuf, +} + +#[derive(Debug, Clone, Copy)] +pub enum McpAgent { + Claude, + Cursor, + Windsurf, +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools.rs b/lib/crates/fabro-mcp-server/src/run_tools.rs new file mode 100644 index 000000000..78c5ed14f --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools.rs @@ -0,0 +1,21 @@ +#![allow( + dead_code, + reason = "MCP DTO fields are consumed by serde and schema generation even when not read directly." +)] + +mod common; +mod create; +mod events; +mod gather; +mod interact; +mod manifest; +mod search; + +pub(crate) use common::{ToolError, error_result, success_result}; +pub(crate) use create::{FabroRunCreateParams, ValidatedCreateRuns, create_runs, create_runs_text}; +pub(crate) use events::{FabroRunEventsParams, ValidatedRunEvents, run_events, run_events_text}; +pub(crate) use gather::{FabroRunGatherParams, ValidatedGatherRuns, gather_runs, gather_runs_text}; +pub(crate) use interact::{ + FabroRunInteractParams, ValidatedInteractRun, interact_run, interact_run_text, +}; +pub(crate) use search::{FabroRunSearchParams, ValidatedSearchRuns, search_runs, search_runs_text}; diff --git a/lib/crates/fabro-mcp-server/src/run_tools/common.rs b/lib/crates/fabro-mcp-server/src/run_tools/common.rs new file mode 100644 index 000000000..d708ed3ad --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/common.rs @@ -0,0 +1,141 @@ +use std::collections::HashMap; + +use chrono::{DateTime, NaiveDate, Utc}; +use fabro_client::Client; +use fabro_types::{Run, RunId, RunStatus}; +use fabro_util::exit::{self, ExitClass}; +use rmcp::model::{CallToolResult, Content}; +use schemars::JsonSchema; +use serde::Serialize; + +#[derive(Debug)] +pub(crate) struct ToolError { + message: String, +} + +impl ToolError { + pub(crate) fn message(message: impl Into) -> Self { + Self { + message: message.into(), + } + } + + pub(crate) fn from_anyhow(err: &anyhow::Error) -> Self { + Self::message(format_tool_error(err)) + } + + pub(crate) fn as_str(&self) -> &str { + &self.message + } +} + +pub(super) type ToolResult = Result; + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct RunSummaryResult { + pub(crate) run_id: String, + pub(crate) workflow_name: String, + pub(crate) workflow_slug: Option, + pub(crate) status: String, + pub(crate) archived: bool, + pub(crate) created_at: String, + pub(crate) started_at: Option, + pub(crate) completed_at: Option, + pub(crate) labels: HashMap, + pub(crate) source_directory: Option, + pub(crate) repo_origin_url: Option, + pub(crate) goal: String, +} + +pub(crate) fn success_result( + value: &T, + text: impl Into, +) -> Result { + let structured_content = serde_json::to_value(value).map_err(|err| { + rmcp::ErrorData::internal_error( + format!("failed to serialize Fabro MCP tool result: {err}"), + None, + ) + })?; + let mut result = CallToolResult::structured(structured_content); + result.content = vec![Content::text(text.into())]; + Ok(result) +} + +pub(crate) fn error_result(err: ToolError) -> CallToolResult { + CallToolResult::error(vec![Content::text(err.message)]) +} + +pub(super) fn validate_len(name: &str, len: usize, min: usize, max: usize) -> ToolResult<()> { + if len < min { + return Err(ToolError::message(format!( + "{name} must contain at least {min} item(s)" + ))); + } + if len > max { + return Err(ToolError::message(format!( + "{name} must contain no more than {max} item(s)" + ))); + } + Ok(()) +} + +pub(super) async fn retrieve_run(client: &Client, run_id: &RunId) -> ToolResult { + client + .retrieve_run(run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err)) +} + +pub(super) fn run_summary_result(run: &Run) -> RunSummaryResult { + RunSummaryResult { + run_id: run.id.to_string(), + workflow_name: run.workflow.name.clone(), + workflow_slug: run.workflow.slug.clone(), + status: run_status_kind(run.lifecycle.status).to_string(), + archived: run.lifecycle.archived, + created_at: run.timestamps.created_at.to_rfc3339(), + started_at: run + .timestamps + .started_at + .map(|timestamp| timestamp.to_rfc3339()), + completed_at: run + .timestamps + .completed_at + .map(|timestamp| timestamp.to_rfc3339()), + labels: run.labels.clone(), + source_directory: run.source_directory.clone(), + repo_origin_url: run + .repository + .as_ref() + .and_then(|repository| repository.origin_url.clone()), + goal: run.goal.clone(), + } +} + +pub(super) fn parse_datetime_filter(name: &str, raw: &str) -> ToolResult> { + if let Ok(timestamp) = DateTime::parse_from_rfc3339(raw) { + return Ok(timestamp.with_timezone(&Utc)); + } + let date = NaiveDate::parse_from_str(raw, "%Y-%m-%d").map_err(|err| { + ToolError::message(format!("{name} must be RFC3339 or YYYY-MM-DD: {err}")) + })?; + let datetime = date + .and_hms_opt(0, 0, 0) + .ok_or_else(|| ToolError::message(format!("{name} contains an invalid date")))?; + Ok(DateTime::from_naive_utc_and_offset(datetime, Utc)) +} + +pub(super) fn run_status_kind(status: RunStatus) -> &'static str { + status.kind().into() +} + +fn format_tool_error(err: &anyhow::Error) -> String { + let mut rendered = format!("{err:#}"); + if exit::exit_class_for(err) == Some(ExitClass::AuthRequired) + && !rendered.contains("fabro auth login") + { + rendered.push_str("\nRun `fabro auth login` to authenticate."); + } + rendered +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/create.rs b/lib/crates/fabro-mcp-server/src/run_tools/create.rs new file mode 100644 index 000000000..56745ff43 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/create.rs @@ -0,0 +1,231 @@ +use std::borrow::Cow; +use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::sync::Arc; + +use fabro_client::Client; +use fabro_types::RunId; +use schemars::{JsonSchema, Schema, SchemaGenerator, json_schema}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use super::common::{ToolError, ToolResult}; +use super::{common, manifest}; + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct FabroRunCreateParams { + pub(crate) runs: Vec, +} + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct CreateRunSpec { + pub(crate) workflow: String, + pub(crate) cwd: Option, + pub(crate) run_id: Option, + pub(crate) goal: Option, + #[serde(default)] + pub(crate) inputs: HashMap, + #[serde(default)] + pub(crate) labels: HashMap, + pub(crate) dry_run: Option, + pub(crate) auto_approve: Option, + pub(crate) model: Option, + pub(crate) provider: Option, + pub(crate) sandbox: Option, + pub(crate) preserve_sandbox: Option, + pub(crate) start: Option, +} + +#[derive(Debug, Deserialize)] +#[serde(transparent)] +pub(crate) struct RunInputValue(Value); + +impl From for RunInputValue { + fn from(value: Value) -> Self { + Self(value) + } +} + +impl RunInputValue { + fn into_inner(self) -> Value { + self.0 + } +} + +impl JsonSchema for RunInputValue { + fn inline_schema() -> bool { + true + } + + fn schema_name() -> Cow<'static, str> { + "RunInputValue".into() + } + + fn json_schema(_: &mut SchemaGenerator) -> Schema { + json_schema!({ + "description": "Run input override value. Inputs are TOML-compatible scalar values: string, boolean, integer, or float.", + "anyOf": [ + { "type": "string" }, + { "type": "boolean" }, + { "type": "integer" }, + { "type": "number" } + ] + }) + } +} + +#[derive(Debug)] +pub(crate) struct ValidatedCreateRuns { + pub(crate) runs: Vec, +} + +#[derive(Debug)] +pub(crate) struct ValidatedCreateRunSpec { + pub(crate) workflow: String, + pub(crate) cwd: Option, + pub(crate) run_id: Option, + pub(crate) goal: Option, + pub(crate) inputs: HashMap, + pub(crate) labels: HashMap, + pub(crate) dry_run: Option, + pub(crate) auto_approve: Option, + pub(crate) model: Option, + pub(crate) provider: Option, + pub(crate) sandbox: Option, + pub(crate) preserve_sandbox: Option, + pub(crate) start: Option, +} + +impl TryFrom for ValidatedCreateRuns { + type Error = ToolError; + + fn try_from(params: FabroRunCreateParams) -> Result { + common::validate_len("runs", params.runs.len(), 1, 50)?; + let runs = params + .runs + .into_iter() + .map(ValidatedCreateRunSpec::try_from) + .collect::, _>>()?; + Ok(Self { runs }) + } +} + +impl TryFrom for ValidatedCreateRunSpec { + type Error = ToolError; + + fn try_from(spec: CreateRunSpec) -> Result { + let run_id = spec + .run_id + .as_deref() + .map(str::parse::) + .transpose() + .map_err(|err| { + ToolError::message(format!("run_id must be a valid Fabro run id: {err}")) + })?; + let inputs = spec + .inputs + .into_iter() + .map(|(key, value)| { + let value = value.into_inner(); + manifest::json_to_toml_value(&key, &value).map(|value| (key, value)) + }) + .collect::>>()?; + Ok(Self { + workflow: spec.workflow, + cwd: spec.cwd, + run_id, + goal: spec.goal, + inputs, + labels: spec.labels, + dry_run: spec.dry_run, + auto_approve: spec.auto_approve, + model: spec.model, + provider: spec.provider, + sandbox: spec.sandbox, + preserve_sandbox: spec.preserve_sandbox, + start: spec.start, + }) + } +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct CreateRunsResult { + pub(crate) runs: Vec, +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct CreatedRunResult { + pub(crate) run_id: String, + pub(crate) workflow: String, + pub(crate) started: bool, + pub(crate) status: String, +} + +pub(crate) async fn create_runs( + client: Arc, + base_cwd: &Path, + user_settings_path: &Path, + params: ValidatedCreateRuns, +) -> ToolResult { + let mut created = Vec::with_capacity(params.runs.len()); + for spec in params.runs { + let cwd = spec.cwd.clone().unwrap_or_else(|| base_cwd.to_path_buf()); + let manifest = manifest::build_mcp_run_manifest(&spec, &cwd, user_settings_path)?; + let run_id = client + .create_run_from_manifest(manifest) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + let started = spec.start.unwrap_or(true); + let summary = if started { + client + .start_run(&run_id, false) + .await + .map_err(|err| ToolError::from_anyhow(&err))? + } else { + client + .retrieve_run(&run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))? + }; + created.push(CreatedRunResult { + run_id: summary.id.to_string(), + workflow: spec.workflow, + started, + status: common::run_status_kind(summary.lifecycle.status).to_string(), + }); + } + Ok(CreateRunsResult { runs: created }) +} + +pub(crate) fn create_runs_text(result: &CreateRunsResult) -> String { + let started = result.runs.iter().filter(|run| run.started).count(); + format!( + "created {} Fabro run(s), started {started}", + result.runs.len() + ) +} + +#[cfg(test)] +mod tests { + use schemars::SchemaGenerator; + use serde_json::json; + + use super::*; + + #[test] + fn run_input_value_schema_allows_only_json_scalars() { + let mut generator = SchemaGenerator::default(); + let schema = RunInputValue::json_schema(&mut generator); + let schema = serde_json::to_value(schema).expect("schema should serialize"); + + assert_eq!( + schema["anyOf"], + json!([ + { "type": "string" }, + { "type": "boolean" }, + { "type": "integer" }, + { "type": "number" }, + ]) + ); + } +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/events.rs b/lib/crates/fabro-mcp-server/src/run_tools/events.rs new file mode 100644 index 000000000..42b7c44cc --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/events.rs @@ -0,0 +1,313 @@ +use std::sync::Arc; + +use chrono::{DateTime, Utc}; +use fabro_client::Client; +use fabro_types::EventEnvelope; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use super::common; +use super::common::{ToolError, ToolResult}; + +#[derive(Debug, Clone, Copy, Deserialize, Serialize, JsonSchema)] +#[serde(rename_all = "snake_case")] +pub(crate) enum RunEventsAction { + List, + Details, + Search, +} + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct FabroRunEventsParams { + pub(crate) action: RunEventsAction, + pub(crate) run_id: String, + pub(crate) event_types: Option>, + pub(crate) categories: Option>, + pub(crate) direction: Option, + pub(crate) created_after: Option, + pub(crate) created_before: Option, + pub(crate) first: Option, + pub(crate) after: Option, + pub(crate) event_ids: Option>, + pub(crate) offset: Option, + pub(crate) limit: Option, + pub(crate) max_content_length: Option, + pub(crate) query: Option, +} + +#[derive(Debug)] +pub(crate) struct ValidatedRunEvents { + pub(crate) raw: FabroRunEventsParams, + pub(crate) descending: bool, + pub(crate) first: usize, + pub(crate) created_after: Option>, + pub(crate) created_before: Option>, +} + +impl TryFrom for ValidatedRunEvents { + type Error = ToolError; + + fn try_from(params: FabroRunEventsParams) -> Result { + if params.run_id.trim().is_empty() { + return Err(ToolError::message("run_id is required")); + } + let first = params.first.or(params.limit).unwrap_or(50); + if first > 200 { + return Err(ToolError::message("first must be <= 200")); + } + let descending = match params.direction.as_deref() { + None | Some("asc") => false, + Some("desc") => true, + Some(_) => return Err(ToolError::message("direction must be `asc` or `desc`")), + }; + let created_after = params + .created_after + .as_deref() + .map(|created_after| common::parse_datetime_filter("created_after", created_after)) + .transpose()?; + let created_before = params + .created_before + .as_deref() + .map(|created_before| common::parse_datetime_filter("created_before", created_before)) + .transpose()?; + if matches!(params.action, RunEventsAction::Details) + && params.event_ids.as_ref().is_none_or(Vec::is_empty) + { + return Err(ToolError::message( + "event_ids is required for details action", + )); + } + if matches!(params.action, RunEventsAction::Search) + && params + .query + .as_deref() + .is_none_or(|query| query.trim().is_empty()) + { + return Err(ToolError::message("query is required for search action")); + } + Ok(Self { + raw: params, + descending, + first, + created_after, + created_before, + }) + } +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct RunEventsResult { + pub(crate) run_id: String, + pub(crate) action: RunEventsAction, + pub(crate) events: Vec, + pub(crate) next_cursor: Option, +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct RunEventResult { + pub(crate) event_id: String, + pub(crate) sequence: u32, + pub(crate) event: Value, + pub(crate) truncated: bool, +} + +pub(crate) async fn run_events( + client: Arc, + params: ValidatedRunEvents, +) -> ToolResult { + let descending = params.descending; + let first = params.first; + let created_after = params.created_after; + let created_before = params.created_before; + let raw = params.raw; + let run_id = client + .resolve_run(&raw.run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))? + .id; + let fetch_after = if descending { None } else { raw.after }; + let mut events = if let Some(limit) = event_fetch_limit(&raw, first) { + client + .list_run_events_until(&run_id, fetch_after, limit) + .await + } else { + client.list_run_events(&run_id, fetch_after, None).await + } + .map_err(|err| ToolError::from_anyhow(&err))?; + if descending { + if let Some(after) = raw.after { + events.retain(|event| event.seq < after); + } + } + filter_events(&mut events, &raw, created_after, created_before); + if descending { + events.reverse(); + } + let offset = raw.offset.unwrap_or(0); + let page = events + .into_iter() + .skip(offset) + .take(first) + .collect::>(); + let max_content_length = raw.max_content_length.unwrap_or(20_000); + let results = page + .iter() + .map(|event| run_event_result(event, max_content_length)) + .collect::>>()?; + let next_cursor = page.last().map(|event| { + if descending { + event.seq + } else { + event.seq.saturating_add(1) + } + }); + + Ok(RunEventsResult { + run_id: run_id.to_string(), + action: raw.action, + events: results, + next_cursor, + }) +} + +pub(crate) fn run_events_text(result: &RunEventsResult) -> String { + format!("returned {} Fabro event(s)", result.events.len()) +} + +fn event_fetch_limit(params: &FabroRunEventsParams, first: usize) -> Option { + let needs_full_scan = params.event_ids.is_some() + || params.event_types.is_some() + || params.categories.is_some() + || params.created_after.is_some() + || params.created_before.is_some() + || params.direction.as_deref() == Some("desc") + || matches!( + params.action, + RunEventsAction::Details | RunEventsAction::Search + ); + if needs_full_scan { + return None; + } + + let requested = first.saturating_add(params.offset.unwrap_or(0)); + Some(requested.max(1)) +} + +fn filter_events( + events: &mut Vec, + params: &FabroRunEventsParams, + created_after: Option>, + created_before: Option>, +) { + if let Some(event_ids) = params.event_ids.as_ref() { + events.retain(|event| event_ids.contains(&event.event.id)); + } + if let Some(event_types) = params.event_types.as_ref() { + events.retain(|event| { + event_types + .iter() + .any(|event_type| event_type == event.event.event_name()) + }); + } + if let Some(categories) = params.categories.as_ref() { + events.retain(|event| { + let category = event + .event + .event_name() + .split('.') + .next() + .unwrap_or_default(); + categories.iter().any(|candidate| candidate == category) + }); + } + if let Some(cutoff) = created_after { + events.retain(|event| event.event.ts >= cutoff); + } + if let Some(cutoff) = created_before { + events.retain(|event| event.event.ts <= cutoff); + } + if matches!(params.action, RunEventsAction::Search) { + if let Some(query) = params.query.as_deref() { + events.retain(|event| { + serde_json::to_string(event).is_ok_and(|serialized| serialized.contains(query)) + }); + } + } +} + +fn run_event_result( + event: &EventEnvelope, + max_content_length: usize, +) -> ToolResult { + let mut serialized = serde_json::to_string(event) + .map_err(|err| ToolError::message(format!("failed to serialize event: {err}")))?; + let truncated = serialized.len() > max_content_length; + let event_value = if truncated { + serialized.truncate(floor_char_boundary(&serialized, max_content_length)); + Value::String(serialized) + } else { + serde_json::to_value(event) + .map_err(|err| ToolError::message(format!("failed to serialize event: {err}")))? + }; + Ok(RunEventResult { + event_id: event.event.id.clone(), + sequence: event.seq, + event: event_value, + truncated, + }) +} + +fn floor_char_boundary(value: &str, max_len: usize) -> usize { + let mut boundary = max_len.min(value.len()); + while !value.is_char_boundary(boundary) { + boundary -= 1; + } + boundary +} + +#[cfg(test)] +mod tests { + use chrono::Utc; + use fabro_types::{EventBody, EventEnvelope, RunEvent, fixtures}; + use serde_json::{Value, json}; + + use super::*; + + #[test] + fn run_event_result_truncates_at_utf8_boundary() { + let event = EventEnvelope { + seq: 1, + event: RunEvent { + id: "evt_utf8".to_string(), + ts: Utc::now(), + run_id: fixtures::RUN_1, + node_id: None, + node_label: None, + stage_id: None, + parallel_group_id: None, + parallel_branch_id: None, + session_id: None, + parent_session_id: None, + tool_call_id: None, + actor: None, + body: EventBody::Unknown { + name: "test.utf8".to_string(), + properties: json!({ "message": "éééé" }), + }, + }, + }; + let serialized = serde_json::to_string(&event).unwrap(); + let first_multibyte = serialized + .find('é') + .expect("serialized event should contain é"); + + let result = run_event_result(&event, first_multibyte + 1).unwrap(); + + assert!(result.truncated); + let Value::String(event_json) = result.event else { + panic!("truncated events should return string payloads"); + }; + assert!(event_json.is_char_boundary(event_json.len())); + } +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/gather.rs b/lib/crates/fabro-mcp-server/src/run_tools/gather.rs new file mode 100644 index 000000000..48e59b33f --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/gather.rs @@ -0,0 +1,109 @@ +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use fabro_client::Client; +use futures::future::try_join_all; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; +use tokio::time; + +use super::common; +use super::common::{RunSummaryResult, ToolError, ToolResult}; + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct FabroRunGatherParams { + pub(crate) run_ids: Vec, + pub(crate) timeout_seconds: Option, + pub(crate) poll_interval_seconds: Option, +} + +#[derive(Debug)] +pub(crate) struct ValidatedGatherRuns { + pub(crate) run_ids: Vec, + pub(crate) timeout_seconds: u64, + pub(crate) poll_interval_seconds: u64, +} + +impl TryFrom for ValidatedGatherRuns { + type Error = ToolError; + + fn try_from(params: FabroRunGatherParams) -> Result { + common::validate_len("run_ids", params.run_ids.len(), 1, 50)?; + if params.timeout_seconds.is_some_and(|timeout| timeout > 600) { + return Err(ToolError::message("timeout_seconds must be <= 600")); + } + if params + .poll_interval_seconds + .is_some_and(|interval| interval < 5) + { + return Err(ToolError::message("poll_interval_seconds must be >= 5")); + } + Ok(Self { + run_ids: params.run_ids, + timeout_seconds: params.timeout_seconds.unwrap_or(300), + poll_interval_seconds: params.poll_interval_seconds.unwrap_or(15), + }) + } +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct GatherRunsResult { + pub(crate) runs: Vec, + pub(crate) timed_out: bool, + pub(crate) elapsed_seconds: u64, +} + +pub(crate) async fn gather_runs( + client: Arc, + params: ValidatedGatherRuns, +) -> ToolResult { + let start = Instant::now(); + let deadline = start + Duration::from_secs(params.timeout_seconds); + let run_ids = try_join_all(params.run_ids.into_iter().map(|selector| { + let client = Arc::clone(&client); + async move { + client + .resolve_run(&selector) + .await + .map(|run| run.id) + .map_err(|err| ToolError::from_anyhow(&err)) + } + })) + .await?; + + loop { + let summaries = try_join_all(run_ids.iter().map(|run_id| { + let client = Arc::clone(&client); + async move { common::retrieve_run(&client, run_id).await } + })) + .await?; + if summaries + .iter() + .all(|run| run.lifecycle.status.is_terminal()) + { + return Ok(GatherRunsResult { + runs: summaries.iter().map(common::run_summary_result).collect(), + timed_out: false, + elapsed_seconds: start.elapsed().as_secs(), + }); + } + let now = Instant::now(); + if now >= deadline { + return Ok(GatherRunsResult { + runs: summaries.iter().map(common::run_summary_result).collect(), + timed_out: true, + elapsed_seconds: start.elapsed().as_secs(), + }); + } + let sleep_for = Duration::from_secs(params.poll_interval_seconds).min(deadline - now); + time::sleep(sleep_for).await; + } +} + +pub(crate) fn gather_runs_text(result: &GatherRunsResult) -> String { + format!( + "gathered {} Fabro run(s), timed_out={}", + result.runs.len(), + result.timed_out + ) +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/interact.rs b/lib/crates/fabro-mcp-server/src/run_tools/interact.rs new file mode 100644 index 000000000..abd6b6cf7 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/interact.rs @@ -0,0 +1,399 @@ +use std::borrow::Cow; +use std::sync::Arc; + +use fabro_api::types; +use fabro_client::Client; +use fabro_types::RunId; +use schemars::{JsonSchema, Schema, SchemaGenerator, json_schema}; +use serde::{Deserialize, Serialize}; +use serde_json::{Value, json}; + +use super::common; +use super::common::{ToolError, ToolResult}; + +#[derive(Debug, Clone, Copy, Deserialize, Serialize, JsonSchema)] +#[serde(rename_all = "snake_case")] +pub(crate) enum RunInteractAction { + Get, + Start, + Message, + Cancel, + Archive, + Unarchive, + GetQuestions, + Answer, +} + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct FabroRunInteractParams { + pub(crate) action: RunInteractAction, + pub(crate) run_id: String, + pub(crate) message: Option, + pub(crate) interrupt: Option, + pub(crate) question_id: Option, + pub(crate) answer: Option, +} + +#[derive(Debug, Deserialize)] +#[serde(transparent)] +pub(crate) struct AnswerValue(Value); + +impl From for AnswerValue { + fn from(value: Value) -> Self { + Self(value) + } +} + +impl AnswerValue { + fn into_inner(self) -> Value { + self.0 + } +} + +impl JsonSchema for AnswerValue { + fn inline_schema() -> bool { + true + } + + fn schema_name() -> Cow<'static, str> { + "AnswerValue".into() + } + + fn json_schema(_: &mut SchemaGenerator) -> Schema { + json_schema!({ + "description": "Answer payload for a pending Fabro question. Use a boolean for yes/no, a string or {\"text\": \"...\"} for freeform text, {\"option\": \"key\"} for a single choice, or {\"options\": [\"key\"]} for multi-select.", + "anyOf": [ + { "type": "boolean" }, + { "type": "string" }, + { + "type": "object", + "properties": { + "option": { "type": "string" } + }, + "required": ["option"], + "additionalProperties": false + }, + { + "type": "object", + "properties": { + "options": { + "type": "array", + "items": { "type": "string" } + } + }, + "required": ["options"], + "additionalProperties": false + }, + { + "type": "object", + "properties": { + "text": { "type": "string" } + }, + "required": ["text"], + "additionalProperties": false + } + ] + }) + } +} + +#[derive(Debug)] +pub(crate) struct ValidatedInteractRun { + pub(crate) run_id: String, + pub(crate) action: ValidatedInteractAction, +} + +#[derive(Debug)] +pub(crate) enum ValidatedInteractAction { + Get, + Start, + Message { + message: String, + interrupt: bool, + }, + Cancel, + Archive, + Unarchive, + GetQuestions, + Answer { + question_id: String, + body: types::SubmitAnswerRequest, + }, +} + +impl ValidatedInteractAction { + fn action(&self) -> RunInteractAction { + match self { + Self::Get => RunInteractAction::Get, + Self::Start => RunInteractAction::Start, + Self::Message { .. } => RunInteractAction::Message, + Self::Cancel => RunInteractAction::Cancel, + Self::Archive => RunInteractAction::Archive, + Self::Unarchive => RunInteractAction::Unarchive, + Self::GetQuestions => RunInteractAction::GetQuestions, + Self::Answer { .. } => RunInteractAction::Answer, + } + } +} + +impl TryFrom for ValidatedInteractRun { + type Error = ToolError; + + fn try_from(params: FabroRunInteractParams) -> Result { + if params.run_id.trim().is_empty() { + return Err(ToolError::message("run_id is required")); + } + let action = match params.action { + RunInteractAction::Get => ValidatedInteractAction::Get, + RunInteractAction::Start => ValidatedInteractAction::Start, + RunInteractAction::Message => { + let Some(message) = params + .message + .as_deref() + .map(str::trim) + .filter(|message| !message.is_empty()) + else { + return Err(ToolError::message("message is required for action message")); + }; + ValidatedInteractAction::Message { + message: message.to_string(), + interrupt: params.interrupt.unwrap_or(false), + } + } + RunInteractAction::Cancel => ValidatedInteractAction::Cancel, + RunInteractAction::Archive => ValidatedInteractAction::Archive, + RunInteractAction::Unarchive => ValidatedInteractAction::Unarchive, + RunInteractAction::GetQuestions => ValidatedInteractAction::GetQuestions, + RunInteractAction::Answer => { + let Some(question_id) = params + .question_id + .as_deref() + .map(str::trim) + .filter(|question_id| !question_id.is_empty()) + else { + return Err(ToolError::message( + "question_id is required for action answer", + )); + }; + let Some(answer) = params.answer else { + return Err(ToolError::message("answer is required for action answer")); + }; + ValidatedInteractAction::Answer { + question_id: question_id.to_string(), + body: answer_to_submit_request(answer.into_inner())?, + } + } + }; + Ok(Self { + run_id: params.run_id.trim().to_string(), + action, + }) + } +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct InteractRunResult { + pub(crate) run_id: String, + pub(crate) action: RunInteractAction, + pub(crate) result: Value, +} + +pub(crate) async fn interact_run( + client: Arc, + params: ValidatedInteractRun, +) -> ToolResult { + let run_id = client + .resolve_run(¶ms.run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))? + .id; + let action = params.action.action(); + let result = match params.action { + ValidatedInteractAction::Get => interact_get(&client, &run_id).await?, + ValidatedInteractAction::Start => { + let summary = client + .start_run(&run_id, false) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "summary": common::run_summary_result(&summary) }) + } + ValidatedInteractAction::Message { message, interrupt } => { + client + .steer_run(&run_id, message.clone(), interrupt) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "message": message, "interrupt": interrupt }) + } + ValidatedInteractAction::Cancel => { + let summary = client + .cancel_run(&run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "summary": common::run_summary_result(&summary) }) + } + ValidatedInteractAction::Archive => { + let summary = client + .archive_run(&run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "summary": common::run_summary_result(&summary) }) + } + ValidatedInteractAction::Unarchive => { + let summary = client + .unarchive_run(&run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "summary": common::run_summary_result(&summary) }) + } + ValidatedInteractAction::GetQuestions => { + let questions = client + .list_run_questions(&run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "questions": questions }) + } + ValidatedInteractAction::Answer { question_id, body } => { + client + .submit_run_answer(&run_id, &question_id, body) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + json!({ "question_id": question_id, "submitted": true }) + } + }; + + Ok(InteractRunResult { + run_id: run_id.to_string(), + action, + result, + }) +} + +pub(crate) fn interact_run_text(result: &InteractRunResult) -> String { + format!( + "completed {:?} for Fabro run {}", + result.action, result.run_id + ) +} + +async fn interact_get(client: &Client, run_id: &RunId) -> ToolResult { + let summary = common::retrieve_run(client, run_id).await?; + let projection = client + .get_run_state(run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err))?; + Ok(json!({ + "summary": common::run_summary_result(&summary), + "projection": projection, + })) +} + +fn answer_to_submit_request(answer: Value) -> ToolResult { + match answer { + Value::Bool(true) => Ok(types::SubmitAnswerYesRequest { + kind: types::SubmitAnswerYesRequestKind::Yes, + } + .into()), + Value::Bool(false) => Ok(types::SubmitAnswerNoRequest { + kind: types::SubmitAnswerNoRequestKind::No, + } + .into()), + Value::String(text) => Ok(text_answer_request(text)), + Value::Object(mut object) => { + if let Some(option) = object.remove("option") { + let option_key = serde_json::from_value::(option).map_err(|err| { + ToolError::message(format!("answer option must be a string: {err}")) + })?; + Ok(types::SubmitAnswerSelectedRequest { + kind: types::SubmitAnswerSelectedRequestKind::Selected, + option_key, + } + .into()) + } else if let Some(options) = object.remove("options") { + let option_keys = + serde_json::from_value::>(options).map_err(|err| { + ToolError::message(format!("answer options must be strings: {err}")) + })?; + Ok(types::SubmitAnswerMultiSelectedRequest { + kind: types::SubmitAnswerMultiSelectedRequestKind::MultiSelected, + option_keys, + } + .into()) + } else if let Some(text) = object.remove("text") { + let text = serde_json::from_value::(text).map_err(|err| { + ToolError::message(format!("answer text must be a string: {err}")) + })?; + Ok(text_answer_request(text)) + } else { + Err(ToolError::message( + "answer object must contain one of: option, options, text", + )) + } + } + other => Err(ToolError::message(format!( + "unsupported answer value: {other}; expected boolean, string, or object", + ))), + } +} + +fn text_answer_request(text: String) -> types::SubmitAnswerRequest { + types::SubmitAnswerTextRequest { + kind: types::SubmitAnswerTextRequestKind::Text, + text, + } + .into() +} + +#[cfg(test)] +mod tests { + use serde_json::json; + + use super::*; + + #[test] + fn answer_payloads_map_to_submit_answer_wire_json() { + let cases = [ + (json!(true), json!({ "kind": "yes" })), + (json!(false), json!({ "kind": "no" })), + (json!("hello"), json!({ "kind": "text", "text": "hello" })), + ( + json!({ "option": "a" }), + json!({ "kind": "selected", "option_key": "a" }), + ), + ( + json!({ "options": ["a", "b"] }), + json!({ "kind": "multi_selected", "option_keys": ["a", "b"] }), + ), + ( + json!({ "text": "hello" }), + json!({ "kind": "text", "text": "hello" }), + ), + ]; + + for (answer, expected) in cases { + let request = answer_to_submit_request(answer).unwrap(); + assert_eq!(serde_json::to_value(request).unwrap(), expected); + } + } + + #[test] + fn unsupported_answer_object_is_rejected() { + let err = answer_to_submit_request(json!({ "value": "yes" })).unwrap_err(); + + assert!(err.as_str().contains("option, options, text")); + } + + #[test] + fn interact_answer_validation_rejects_unsupported_json_before_api_calls() { + let err = ValidatedInteractRun::try_from(FabroRunInteractParams { + action: RunInteractAction::Answer, + run_id: "run_123".to_string(), + message: None, + interrupt: None, + question_id: Some("question-1".to_string()), + answer: Some(json!({ "value": "yes" }).into()), + }) + .unwrap_err(); + + assert!(err.as_str().contains("option, options, text")); + } +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/manifest.rs b/lib/crates/fabro-mcp-server/src/run_tools/manifest.rs new file mode 100644 index 000000000..581bf21d2 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/manifest.rs @@ -0,0 +1,178 @@ +use std::path::{Path, PathBuf}; + +use fabro_api::types; +use fabro_config::{CliLayer, RunLayer}; +use fabro_manifest::{self, ManifestBuildInput, RunOverrideInput}; +use fabro_server::manifest_validation; +use serde_json::Value; + +use super::common::{ToolError, ToolResult}; +use super::create::ValidatedCreateRunSpec; + +pub(super) fn build_mcp_run_manifest( + spec: &ValidatedCreateRunSpec, + cwd: &Path, + user_settings_path: &Path, +) -> ToolResult { + let built = fabro_manifest::build_run_manifest(ManifestBuildInput { + workflow: PathBuf::from(&spec.workflow), + cwd: cwd.to_path_buf(), + run_overrides: mcp_run_overrides(spec), + cli_overrides: Some(CliLayer::default()), + input_overrides: spec.inputs.clone(), + args: mcp_manifest_args(spec), + run_id: spec.run_id, + user_settings_path: Some(user_settings_path.to_path_buf()), + }) + .map_err(|err| ToolError::from_anyhow(&err))?; + let validation = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest) + .map_err(|err| ToolError::from_anyhow(&err))?; + if !validation.ok { + return Err(ToolError::message("workflow manifest validation failed")); + } + Ok(built.manifest) +} + +pub(super) fn json_to_toml_value(key: &str, value: &Value) -> ToolResult { + match value { + Value::Null => Err(ToolError::message(format!( + "input `{key}` cannot be null; use a string, boolean, or number" + ))), + Value::Bool(value) => Ok(toml::Value::Boolean(*value)), + Value::Number(value) => { + if let Some(integer) = value.as_i64() { + Ok(toml::Value::Integer(integer)) + } else if let Some(float) = value.as_f64() { + Ok(toml::Value::Float(float)) + } else { + Err(ToolError::message(format!( + "input `{key}` contains a number outside TOML's supported range" + ))) + } + } + Value::String(value) => Ok(toml::Value::String(value.clone())), + Value::Array(_) => Err(ToolError::message(format!( + "input `{key}` does not support array values; use a string, boolean, or number", + ))), + Value::Object(_) => Err(ToolError::message(format!( + "input `{key}` does not support object values; use a string, boolean, or number", + ))), + } +} + +fn mcp_manifest_args(spec: &ValidatedCreateRunSpec) -> Option { + let mut input = spec + .inputs + .iter() + .map(|(key, value)| format!("{key}={value}")) + .collect::>(); + input.sort(); + let mut label = spec + .labels + .iter() + .map(|(key, value)| format!("{key}={value}")) + .collect::>(); + label.sort(); + let payload = types::ManifestArgs { + auto_approve: spec.auto_approve.filter(|value| *value), + docker_image: None, + dry_run: spec.dry_run.filter(|value| *value), + input, + label, + model: spec.model.clone(), + preserve_sandbox: spec.preserve_sandbox.filter(|value| *value), + provider: spec.provider.clone(), + sandbox: spec.sandbox.clone(), + verbose: None, + }; + (!fabro_manifest::manifest_args_is_empty(&payload)).then_some(payload) +} + +fn mcp_run_overrides(spec: &ValidatedCreateRunSpec) -> Option { + fabro_manifest::build_sparse_run_overrides(RunOverrideInput { + goal: spec.goal.as_deref(), + model: spec.model.as_deref(), + provider: spec.provider.as_deref(), + sandbox: spec.sandbox.as_deref(), + docker_image: None, + preserve_sandbox: spec.preserve_sandbox, + dry_run: spec.dry_run, + auto_approve: spec.auto_approve, + labels: spec.labels.clone(), + }) +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use serde_json::{Value, json}; + + use super::super::create::CreateRunSpec; + use super::*; + + #[test] + fn json_inputs_convert_scalar_values_to_toml_values() { + let cases = [ + (json!("hello"), toml::Value::String("hello".to_string())), + (json!(true), toml::Value::Boolean(true)), + (json!(42), toml::Value::Integer(42)), + (json!(0.5), toml::Value::Float(0.5)), + ]; + + for (json, expected) in cases { + assert_eq!(json_to_toml_value("input", &json).unwrap(), expected); + } + } + + #[test] + fn json_input_arrays_and_objects_are_rejected() { + let array_err = json_to_toml_value("matrix", &json!(["a", 1])).unwrap_err(); + assert_eq!( + array_err.as_str(), + "input `matrix` does not support array values; use a string, boolean, or number", + ); + + let object_err = json_to_toml_value("settings", &json!({ "enabled": true })).unwrap_err(); + assert_eq!( + object_err.as_str(), + "input `settings` does not support object values; use a string, boolean, or number", + ); + } + + #[test] + fn json_input_null_is_rejected_with_key_name() { + let err = json_to_toml_value("goal", &Value::Null).unwrap_err(); + + assert_eq!( + err.as_str(), + "input `goal` cannot be null; use a string, boolean, or number", + ); + } + + #[test] + fn mcp_manifest_args_preserve_input_provenance() { + let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec { + workflow: "simple".to_string(), + run_id: None, + cwd: None, + goal: None, + inputs: HashMap::from([ + ("count".to_string(), json!(3).into()), + ("decision".to_string(), json!("approve").into()), + ]), + labels: HashMap::new(), + model: None, + provider: None, + sandbox: None, + dry_run: None, + auto_approve: None, + preserve_sandbox: None, + start: None, + }) + .expect("create spec should validate"); + let args = mcp_manifest_args(&spec).expect("input args should be present"); + + assert_eq!(args.input, vec![r"count=3", r#"decision="approve""#]); + } +} diff --git a/lib/crates/fabro-mcp-server/src/run_tools/search.rs b/lib/crates/fabro-mcp-server/src/run_tools/search.rs new file mode 100644 index 000000000..0a5b2bb15 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/run_tools/search.rs @@ -0,0 +1,378 @@ +use std::collections::HashMap; +use std::sync::Arc; + +use fabro_client::Client; +use fabro_types::{Run, RunStatusKind}; +use futures::future::try_join_all; +use schemars::JsonSchema; +use serde::{Deserialize, Serialize}; + +use super::common; +use super::common::{RunSummaryResult, ToolError, ToolResult}; + +const SEARCH_GOAL_PREVIEW_CHARS: usize = 240; + +#[derive(Debug, Deserialize, JsonSchema)] +pub(crate) struct FabroRunSearchParams { + pub(crate) run_ids: Option>, + pub(crate) workflow: Option, + pub(crate) labels: Option>, + pub(crate) status: Option>, + pub(crate) archived: Option, + pub(crate) created_after: Option, + pub(crate) created_before: Option, + pub(crate) first: Option, + pub(crate) after: Option, +} + +#[derive(Debug)] +pub(crate) struct ValidatedSearchRuns { + pub(crate) raw: FabroRunSearchParams, + pub(crate) status: Option>, +} + +impl TryFrom for ValidatedSearchRuns { + type Error = ToolError; + + fn try_from(params: FabroRunSearchParams) -> Result { + if params.first.is_some_and(|first| first > 100) { + return Err(ToolError::message("first must be <= 100")); + } + if let Some(run_ids) = params.run_ids.as_ref() { + common::validate_len("run_ids", run_ids.len(), 1, 100)?; + } + let status = params + .status + .as_ref() + .map(|statuses| { + statuses + .iter() + .map(|status| { + status.parse::().map_err(|_| { + ToolError::message(format!("unknown run status `{status}`")) + }) + }) + .collect::>>() + }) + .transpose()?; + if let Some(created_after) = params.created_after.as_deref() { + common::parse_datetime_filter("created_after", created_after)?; + } + if let Some(created_before) = params.created_before.as_deref() { + common::parse_datetime_filter("created_before", created_before)?; + } + Ok(Self { + raw: params, + status, + }) + } +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct SearchRunsResult { + pub(crate) runs: Vec, + pub(crate) next_cursor: Option, +} + +#[derive(Debug, Serialize, JsonSchema)] +pub(crate) struct SearchRunSummaryResult { + pub(crate) run_id: String, + pub(crate) workflow_name: String, + pub(crate) workflow_slug: Option, + pub(crate) status: String, + pub(crate) archived: bool, + pub(crate) created_at: String, + pub(crate) started_at: Option, + pub(crate) completed_at: Option, + pub(crate) labels: HashMap, + pub(crate) source_directory: Option, + pub(crate) repo_origin_url: Option, + pub(crate) goal_preview: String, + pub(crate) goal_truncated: bool, +} + +pub(crate) async fn search_runs( + client: Arc, + params: ValidatedSearchRuns, +) -> ToolResult { + let status = params.status; + let raw = params.raw; + let runs = if let Some(run_ids) = raw.run_ids.as_ref() { + resolve_requested_runs(&client, run_ids).await? + } else { + client + .list_store_runs() + .await + .map_err(|err| ToolError::from_anyhow(&err))? + }; + let page = filter_sort_and_page_runs(runs, &raw, status.as_deref())?; + + Ok(SearchRunsResult { + runs: page.runs.iter().map(search_run_summary_result).collect(), + next_cursor: page.next_cursor, + }) +} + +fn search_run_summary_result(run: &Run) -> SearchRunSummaryResult { + let RunSummaryResult { + run_id, + workflow_name, + workflow_slug, + status, + archived, + created_at, + started_at, + completed_at, + labels, + source_directory, + repo_origin_url, + goal, + } = common::run_summary_result(run); + let (goal_preview, goal_truncated) = goal_preview(&goal); + + SearchRunSummaryResult { + run_id, + workflow_name, + workflow_slug, + status, + archived, + created_at, + started_at, + completed_at, + labels, + source_directory, + repo_origin_url, + goal_preview, + goal_truncated, + } +} + +fn goal_preview(goal: &str) -> (String, bool) { + let mut chars = goal.chars(); + let mut preview = chars + .by_ref() + .take(SEARCH_GOAL_PREVIEW_CHARS) + .collect::(); + let truncated = chars.next().is_some(); + if truncated { + preview.push_str("..."); + } + (preview, truncated) +} + +struct RunSearchPage { + runs: Vec, + next_cursor: Option, +} + +fn filter_sort_and_page_runs( + mut runs: Vec, + raw: &FabroRunSearchParams, + status: Option<&[RunStatusKind]>, +) -> ToolResult { + if let Some(workflow) = raw.workflow.as_deref() { + runs.retain(|run| { + run.workflow.name == workflow || run.workflow.slug.as_deref() == Some(workflow) + }); + } + if let Some(labels) = raw.labels.as_ref() { + runs.retain(|run| { + labels + .iter() + .all(|(key, value)| run.labels.get(key) == Some(value)) + }); + } + if let Some(status) = status { + runs.retain(|run| { + status + .iter() + .any(|status| *status == run.lifecycle.status.kind()) + }); + } + let archived = raw.archived.unwrap_or(false); + runs.retain(|run| run.lifecycle.archived == archived); + if let Some(created_after) = raw.created_after.as_deref() { + let cutoff = common::parse_datetime_filter("created_after", created_after)?; + runs.retain(|run| run.timestamps.created_at >= cutoff); + } + if let Some(created_before) = raw.created_before.as_deref() { + let cutoff = common::parse_datetime_filter("created_before", created_before)?; + runs.retain(|run| run.timestamps.created_at <= cutoff); + } + + runs.sort_by(|a, b| { + let a_sort_time = a.timestamps.started_at.unwrap_or(a.timestamps.created_at); + let b_sort_time = b.timestamps.started_at.unwrap_or(b.timestamps.created_at); + b_sort_time.cmp(&a_sort_time).then_with(|| b.id.cmp(&a.id)) + }); + + if let Some(after) = raw.after.as_deref() { + if let Some(position) = runs.iter().position(|run| run.id.to_string() == after) { + runs = runs.into_iter().skip(position + 1).collect(); + } + } + + let first = raw.first.unwrap_or(20).min(100); + let has_more = runs.len() > first; + let page = runs.into_iter().take(first).collect::>(); + let next_cursor = has_more + .then(|| page.last().map(|run| run.id.to_string())) + .flatten(); + Ok(RunSearchPage { + runs: page, + next_cursor, + }) +} + +pub(crate) fn search_runs_text(result: &SearchRunsResult) -> String { + format!("found {} Fabro run(s)", result.runs.len()) +} + +async fn resolve_requested_runs(client: &Arc, run_ids: &[String]) -> ToolResult> { + let runs = try_join_all(run_ids.iter().map(|run_id| { + let client = Arc::clone(client); + async move { + client + .resolve_run(run_id) + .await + .map_err(|err| ToolError::from_anyhow(&err)) + } + })) + .await?; + + let mut unique = HashMap::new(); + for run in runs { + unique.entry(run.id).or_insert(run); + } + Ok(unique.into_values().collect()) +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use chrono::{TimeZone, Utc}; + use fabro_types::{RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef}; + + use super::*; + + #[test] + fn cursor_is_applied_after_filters() { + let matching_newer = run("01KRBZW5C00000000000000001", "keep", 30); + let unrelated_cursor = run("01KRBZW4DW0000000000000002", "skip", 20); + let matching_older = run("01KRBZW3EF0000000000000003", "keep", 10); + + let result = filter_sort_and_page_runs( + vec![ + matching_older.clone(), + unrelated_cursor.clone(), + matching_newer.clone(), + ], + &FabroRunSearchParams { + run_ids: None, + workflow: None, + labels: Some(HashMap::from([("group".to_string(), "keep".to_string())])), + status: None, + archived: None, + created_after: None, + created_before: None, + first: Some(10), + after: Some(unrelated_cursor.id.to_string()), + }, + None, + ) + .expect("filtering should succeed"); + + let ids = result.runs.iter().map(|run| run.id).collect::>(); + assert_eq!(ids, vec![matching_newer.id, matching_older.id]); + } + + #[test] + fn omitted_archived_filter_hides_archived_runs_by_default() { + let active = run("01KRBZW5C00000000000000001", "keep", 30); + let archived = archived_run("01KRBZW4DW0000000000000002", "keep", 20); + + let result = filter_sort_and_page_runs( + vec![archived.clone(), active.clone()], + &FabroRunSearchParams { + run_ids: None, + workflow: None, + labels: None, + status: None, + archived: None, + created_after: None, + created_before: None, + first: Some(10), + after: None, + }, + None, + ) + .expect("filtering should succeed"); + + let ids = result.runs.iter().map(|run| run.id).collect::>(); + assert_eq!(ids, vec![active.id]); + } + + #[test] + fn search_summary_uses_bounded_goal_preview() { + let mut run = run("01KRBZW5C00000000000000001", "keep", 30); + run.goal = format!("{}tail-marker", "a".repeat(300)); + + let summary = search_run_summary_result(&run); + + assert!(summary.goal_truncated); + assert!(summary.goal_preview.len() < run.goal.len()); + assert!(!summary.goal_preview.contains("tail-marker")); + } + + fn run(id: &str, group: &str, seconds: u32) -> Run { + run_with_archived(id, group, seconds, false) + } + + fn archived_run(id: &str, group: &str, seconds: u32) -> Run { + run_with_archived(id, group, seconds, true) + } + + fn run_with_archived(id: &str, group: &str, seconds: u32, archived: bool) -> Run { + let created_at = Utc.with_ymd_and_hms(2026, 5, 11, 12, 0, seconds).unwrap(); + Run { + id: id.parse().expect("test run id should parse"), + title: "test".to_string(), + goal: "test".to_string(), + workflow: WorkflowRef { + slug: Some("simple".to_string()), + name: "Simple".to_string(), + }, + automation: None, + repository: None, + created_by: None, + origin: RunOrigin::default(), + labels: HashMap::from([("group".to_string(), group.to_string())]), + lifecycle: RunLifecycle { + status: RunStatus::Submitted, + pending_control: None, + queue_position: None, + error: None, + archived, + archived_at: None, + }, + sandbox: None, + models: Vec::new(), + source_directory: None, + timestamps: RunTimestamps { + created_at, + started_at: None, + last_event_at: None, + completed_at: None, + duration_ms: None, + elapsed_secs: None, + }, + billing: None, + diff: None, + pull_request: None, + current_question: None, + superseded_by: None, + links: RunLinks { web: None }, + } + } +} diff --git a/lib/crates/fabro-mcp-server/src/server.rs b/lib/crates/fabro-mcp-server/src/server.rs new file mode 100644 index 000000000..e82b2e404 --- /dev/null +++ b/lib/crates/fabro-mcp-server/src/server.rs @@ -0,0 +1,171 @@ +use std::path::PathBuf; +use std::sync::Arc; + +use anyhow::Result; +use fabro_client::Client; +use rmcp::handler::server::router::tool::ToolRouter; +use rmcp::handler::server::wrapper::Parameters; +use rmcp::model::{CallToolResult, ServerCapabilities, ServerInfo}; +use rmcp::transport::stdio; +use rmcp::{ErrorData, ServerHandler, serve_server, tool, tool_handler, tool_router}; +use tokio::sync::OnceCell; + +use crate::{FabroMcpServerSettings, run_tools}; + +#[derive(Clone)] +pub(crate) struct FabroMcpServer { + settings: Arc, + client: Arc>>, + cwd: PathBuf, + tool_router: ToolRouter, +} + +pub async fn start(settings: FabroMcpServerSettings) -> Result<()> { + let server = FabroMcpServer::new(Arc::new(settings)); + let service = serve_server(server, stdio()).await?; + service.waiting().await?; + Ok(()) +} + +#[tool_handler(router = self.tool_router)] +impl ServerHandler for FabroMcpServer { + fn get_info(&self) -> ServerInfo { + ServerInfo::new(ServerCapabilities::builder().enable_tools().build()) + .with_instructions("Use these tools to create, inspect, control, wait for, and read events from Fabro workflow runs.") + } +} + +#[tool_router(router = tool_router)] +impl FabroMcpServer { + pub(crate) fn new(settings: Arc) -> Self { + let cwd = settings.cwd.clone(); + Self { + settings, + client: Arc::new(OnceCell::new()), + cwd, + tool_router: Self::tool_router(), + } + } + + #[tool( + name = "fabro_run_create", + description = "Create one or more Fabro workflow runs, starting them by default." + )] + async fn fabro_run_create( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedCreateRuns::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::create_runs(client, &self.cwd, &self.settings.config_path, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::create_runs_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } + + #[tool( + name = "fabro_run_search", + description = "Search Fabro workflow runs by id, workflow, labels, status, archival state, and creation time." + )] + async fn fabro_run_search( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedSearchRuns::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::search_runs(client, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::search_runs_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } + + #[tool( + name = "fabro_run_interact", + description = "Get, start, message, cancel, archive, unarchive, inspect questions, or answer a Fabro run." + )] + async fn fabro_run_interact( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedInteractRun::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::interact_run(client, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::interact_run_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } + + #[tool( + name = "fabro_run_gather", + description = "Wait for Fabro runs to reach terminal states, returning current state on timeout." + )] + async fn fabro_run_gather( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedGatherRuns::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::gather_runs(client, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::gather_runs_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } + + #[tool( + name = "fabro_run_events", + description = "List, inspect, or search stored events for a Fabro workflow run." + )] + async fn fabro_run_events( + &self, + params: Parameters, + ) -> Result { + let params = match run_tools::ValidatedRunEvents::try_from(params.0) { + Ok(params) => params, + Err(err) => return Ok(run_tools::error_result(err)), + }; + let client = match self.client().await { + Ok(client) => client, + Err(err) => return Ok(run_tools::error_result(err)), + }; + match run_tools::run_events(client, params).await { + Ok(result) => run_tools::success_result(&result, run_tools::run_events_text(&result)), + Err(err) => Ok(run_tools::error_result(err)), + } + } + + async fn client(&self) -> Result, run_tools::ToolError> { + self.client + .get_or_try_init(|| async { + (self.settings.client_factory)() + .await + .map(Arc::new) + .map_err(|err| run_tools::ToolError::from_anyhow(&err)) + }) + .await + .map(Arc::clone) + } +} diff --git a/lib/crates/fabro-mcp/src/client.rs b/lib/crates/fabro-mcp/src/client.rs index 1c4e537a6..1775affff 100644 --- a/lib/crates/fabro-mcp/src/client.rs +++ b/lib/crates/fabro-mcp/src/client.rs @@ -22,6 +22,8 @@ enum ClientState { Connecting(Option), /// Handshake complete, ready for tool calls. Ready(Arc>), + /// Connection was explicitly closed. + Closed, } enum PendingTransport { @@ -51,9 +53,15 @@ impl McpClient { .stderr(Stdio::piped()) .kill_on_drop(true); + if config.clear_env { + cmd.env_clear(); + } if !env.is_empty() { cmd.envs(env); } + if let Some(current_dir) = config.current_dir.as_ref() { + cmd.current_dir(current_dir); + } #[cfg(unix)] cmd.process_group(0); @@ -116,6 +124,7 @@ impl McpClient { .take() .ok_or_else(|| anyhow!("client already initializing"))?, ClientState::Ready(_) => return Err(anyhow!("client already initialized")), + ClientState::Closed => return Err(anyhow!("MCP client is shut down")), }; // Drop the lock before the blocking handshake @@ -243,11 +252,38 @@ impl McpClient { Ok(result) } + pub async fn shutdown(self) -> Result<()> { + let service = { + let mut guard = self.state.lock().await; + match std::mem::replace(&mut *guard, ClientState::Closed) { + ClientState::Connecting(_) | ClientState::Closed => None, + ClientState::Ready(service) => Some(service), + } + }; + + if let Some(service) = service { + match Arc::try_unwrap(service) { + Ok(mut service) => { + service + .close_with_timeout(Duration::from_secs(2)) + .await + .context("failed to shut down MCP client")?; + } + Err(service) => { + service.cancellation_token().cancel(); + } + } + } + + Ok(()) + } + async fn service(&self) -> Result>> { let guard = self.state.lock().await; match &*guard { ClientState::Ready(service) => Ok(Arc::clone(service)), ClientState::Connecting(_) => Err(anyhow!("MCP client not initialized")), + ClientState::Closed => Err(anyhow!("MCP client is shut down")), } } } diff --git a/lib/crates/fabro-mcp/tests/stdio_integration.rs b/lib/crates/fabro-mcp/tests/stdio_integration.rs index 8da5c24d6..bd0b2ed73 100644 --- a/lib/crates/fabro-mcp/tests/stdio_integration.rs +++ b/lib/crates/fabro-mcp/tests/stdio_integration.rs @@ -13,6 +13,8 @@ fn test_server_config() -> McpServerSettings { command: vec!["python3".into(), test_server], env: HashMap::new(), }, + current_dir: None, + clear_env: false, startup_timeout_secs: 10, tool_timeout_secs: 30, } @@ -30,6 +32,78 @@ async fn stdio_client_initialize_and_list_tools() { assert_eq!(tools[0].1, "Echo back the message"); } +#[tokio::test] +#[expect( + clippy::disallowed_methods, + reason = "stdio integration test stages a local process cwd and inherits PATH for python3 lookup" +)] +async fn stdio_client_uses_configured_cwd_and_exact_env() { + let test_server = format!("{}/tests/test_mcp_server.py", env!("CARGO_MANIFEST_DIR")); + let temp_dir = std::env::temp_dir().join(format!( + "fabro-mcp-stdio-{}-{}", + std::process::id(), + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap() + .as_nanos() + )); + std::fs::create_dir(&temp_dir).unwrap(); + let canonical_temp_dir = std::fs::canonicalize(&temp_dir).unwrap(); + let mut env = HashMap::new(); + env.insert( + "PATH".to_string(), + std::env::var("PATH").expect("PATH should be set for python3 lookup"), + ); + env.insert("FABRO_MCP_TEST_SENTINEL".to_string(), "fixture".to_string()); + let config = McpServerSettings { + name: "test-echo".into(), + transport: McpTransport::Stdio { + command: vec!["python3".into(), test_server], + env, + }, + current_dir: Some(canonical_temp_dir.clone()), + clear_env: true, + startup_timeout_secs: 10, + tool_timeout_secs: 30, + }; + let client = McpClient::new(&config).unwrap(); + client.initialize(config.startup_timeout()).await.unwrap(); + + let cwd = client + .call_tool( + "echo", + serde_json::json!({"message": "__cwd__"}), + Duration::from_secs(5), + ) + .await + .unwrap(); + assert_eq!( + call_result_to_string(&cwd).unwrap(), + canonical_temp_dir.display().to_string() + ); + let sentinel = client + .call_tool( + "echo", + serde_json::json!({"message": "__env:FABRO_MCP_TEST_SENTINEL__"}), + Duration::from_secs(5), + ) + .await + .unwrap(); + assert_eq!(call_result_to_string(&sentinel).unwrap(), "fixture"); + let home = client + .call_tool( + "echo", + serde_json::json!({"message": "__env:HOME__"}), + Duration::from_secs(5), + ) + .await + .unwrap(); + assert_eq!(call_result_to_string(&home).unwrap(), ""); + + client.shutdown().await.unwrap(); + std::fs::remove_dir(&temp_dir).unwrap(); +} + #[tokio::test] async fn stdio_client_call_tool_echo() { let config = test_server_config(); diff --git a/lib/crates/fabro-mcp/tests/test_mcp_server.py b/lib/crates/fabro-mcp/tests/test_mcp_server.py index 619d0f365..63f95b689 100644 --- a/lib/crates/fabro-mcp/tests/test_mcp_server.py +++ b/lib/crates/fabro-mcp/tests/test_mcp_server.py @@ -5,6 +5,7 @@ Speaks JSON-RPC 2.0 over stdin/stdout per the MCP specification. Exposes a single tool: echo(message) -> message. """ import json +import os import sys SERVER_INFO = { @@ -53,6 +54,11 @@ def handle_request(req): arguments = params.get("arguments", {}) if tool_name == "echo": msg = arguments.get("message", "") + if msg == "__cwd__": + msg = os.getcwd() + elif msg.startswith("__env:") and msg.endswith("__"): + key = msg[len("__env:") : -len("__")] + msg = os.environ.get(key, "") return { "jsonrpc": "2.0", "id": req_id, diff --git a/lib/crates/fabro-sandbox/Cargo.toml b/lib/crates/fabro-sandbox/Cargo.toml index b7d3f51f5..0f6c3f874 100644 --- a/lib/crates/fabro-sandbox/Cargo.toml +++ b/lib/crates/fabro-sandbox/Cargo.toml @@ -24,7 +24,7 @@ anyhow.workspace = true async-trait.workspace = true thiserror.workspace = true tokio.workspace = true -tokio-util.workspace = true +tokio-util = { workspace = true, features = ["compat"] } serde.workspace = true serde_json.workspace = true strum.workspace = true diff --git a/lib/crates/fabro-sandbox/src/daytona/mod.rs b/lib/crates/fabro-sandbox/src/daytona/mod.rs index 572b3f422..646cae16c 100644 --- a/lib/crates/fabro-sandbox/src/daytona/mod.rs +++ b/lib/crates/fabro-sandbox/src/daytona/mod.rs @@ -29,7 +29,7 @@ use crate::redact::redact_auth_url; use crate::sandbox::{optional_timeout, resolve_path}; use crate::{ CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox, - SandboxEvent, SandboxEventCallback, format_lines_numbered, shell_quote, + SandboxEvent, SandboxEventCallback, StdioProcess, format_lines_numbered, shell_quote, }; const WORKING_DIRECTORY: &str = "/home/daytona/workspace"; @@ -1535,6 +1535,18 @@ impl Sandbox for DaytonaSandbox { }) } + async fn spawn_stdio_process( + &self, + _command: &str, + _working_dir: Option<&str>, + _env_vars: Option<&HashMap>, + _cancel_token: Option, + ) -> crate::Result { + Err(crate::Error::message( + "ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet", + )) + } + async fn grep( &self, pattern: &str, diff --git a/lib/crates/fabro-sandbox/src/details.rs b/lib/crates/fabro-sandbox/src/details.rs index 2abb93cad..317b8ee6d 100644 --- a/lib/crates/fabro-sandbox/src/details.rs +++ b/lib/crates/fabro-sandbox/src/details.rs @@ -1,6 +1,7 @@ use std::collections::BTreeMap; use anyhow::Result; +#[cfg(any(feature = "docker", feature = "daytona"))] use chrono::{DateTime, Utc}; use fabro_types::{ RunId, RunSandbox, SandboxDetails, SandboxProvider, SandboxResources, SandboxState, @@ -55,6 +56,7 @@ fn local_details(record: &RunSandbox) -> SandboxDetails { } } +#[cfg(any(feature = "docker", feature = "daytona"))] fn parse_rfc3339_utc(value: &str) -> Option> { DateTime::parse_from_rfc3339(value) .ok() diff --git a/lib/crates/fabro-sandbox/src/docker.rs b/lib/crates/fabro-sandbox/src/docker.rs index b465a2eb6..d85bc138a 100644 --- a/lib/crates/fabro-sandbox/src/docker.rs +++ b/lib/crates/fabro-sandbox/src/docker.rs @@ -1,7 +1,8 @@ use std::collections::HashMap; use std::fmt::Write as _; use std::io::Cursor; -use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::time::{Instant, SystemTime, UNIX_EPOCH}; use async_trait::async_trait; @@ -12,23 +13,25 @@ use bollard::container::{ UploadToContainerOptions, }; use bollard::errors::Error as DockerError; -use bollard::exec::{CreateExecOptions, StartExecResults}; +use bollard::exec::{CreateExecOptions, StartExecOptions, StartExecResults}; use bollard::image::CreateImageOptions; use bollard::models::HostConfig; use fabro_github::GitHubCredentials; use fabro_types::{CommandOutputStream, CommandTermination, RunId}; use fabro_util::time::elapsed_ms; use futures::StreamExt; -use tokio::sync::OnceCell; +use tokio::io::{AsyncWriteExt, duplex}; +use tokio::sync::{Mutex as TokioMutex, Notify, OnceCell}; use tokio::{fs, time}; use tokio_util::sync::CancellationToken; use crate::clone_source::{self, CloneDecision, EmptyWorkspaceReason}; use crate::redact::redact_auth_url; -use crate::sandbox::{optional_timeout, resolve_path}; +use crate::sandbox::{StdioProcessControl, optional_timeout, resolve_path}; use crate::{ - CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox, - SandboxEvent, SandboxEventCallback, format_lines_numbered, shell_quote, + CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult, + ExecStreamingResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback, StderrCollector, + StdioProcess, StdioProcessHandle, StdioProcessTermination, format_lines_numbered, shell_quote, }; const WORKING_DIRECTORY: &str = "/workspace"; @@ -216,17 +219,15 @@ impl DockerSandbox { ..Default::default() }; - let exec_instance = self - .docker - .create_exec(container_id, exec_opts) - .await - .map_err(|e| crate::Error::context("Failed to create exec", e))?; - - let start_result = self - .docker - .start_exec(&exec_instance.id, None) - .await - .map_err(|e| crate::Error::context("Failed to start exec", e))?; + let (exec_id, start_result) = create_and_start_exec( + &self.docker, + container_id, + exec_opts, + None, + "Failed to create exec", + "Failed to start exec", + ) + .await?; let mut stdout = String::new(); let mut stderr = String::new(); @@ -250,7 +251,7 @@ impl DockerSandbox { let inspect = self .docker - .inspect_exec(&exec_instance.id) + .inspect_exec(&exec_id) .await .map_err(|e| crate::Error::context("Failed to inspect exec", e))?; @@ -278,15 +279,15 @@ impl DockerSandbox { ..Default::default() }; - let exec_instance = docker - .create_exec(&container_id, exec_opts) - .await - .map_err(|e| crate::Error::context("Failed to create exec", e))?; - - let start_result = docker - .start_exec(&exec_instance.id, None) - .await - .map_err(|e| crate::Error::context("Failed to start exec", e))?; + let (exec_id, start_result) = create_and_start_exec( + &docker, + &container_id, + exec_opts, + None, + "Failed to create exec", + "Failed to start exec", + ) + .await?; let mut stdout = Vec::new(); let mut stderr = Vec::new(); @@ -311,7 +312,7 @@ impl DockerSandbox { } let inspect = docker - .inspect_exec(&exec_instance.id) + .inspect_exec(&exec_id) .await .map_err(|e| crate::Error::context("Failed to inspect exec", e))?; @@ -451,20 +452,7 @@ impl DockerSandbox { } async fn request_docker_exec_stop(&self, stop_file: &str) -> crate::Result<()> { - let command = format!("touch {}", shell_quote(stop_file)); - let (stdout, stderr, exit_code) = self - .docker_exec( - vec!["/bin/bash".to_string(), "-lc".to_string(), command.clone()], - Some("/"), - None, - ) - .await?; - if exit_code != 0 { - return Err(crate::Error::message(format!( - "Failed to request Docker exec stop (exit {exit_code}): {stderr}{stdout}" - ))); - } - Ok(()) + request_docker_exec_stop_with(&self.docker, self.container_id()?, stop_file).await } async fn ensure_image(&self) -> crate::Result { @@ -769,6 +757,7 @@ fn docker_controlled_shell_command(command: &str, stop_file: &str, pid_file: &st stop_file={stop_file}; \ pid_file={pid_file}; \ user_command={command}; \ +exec 3<&0; \ rm -f \"$pid_file\"; \ if [ -e \"$stop_file\" ]; then \ rm -f \"$stop_file\" \"$pid_file\"; \ @@ -783,11 +772,12 @@ fi; \ kill -KILL \"-$child\" 2>/dev/null || kill -KILL \"$child\" 2>/dev/null || true; \ ) & watcher=$!; \ if command -v setsid >/dev/null 2>&1; then \ - setsid /bin/bash -lc \"$user_command\" & \ + setsid /bin/bash -lc \"$user_command\" <&3 & \ else \ - /bin/bash -lc \"$user_command\" & \ + /bin/bash -lc \"$user_command\" <&3 & \ fi; \ child=$!; \ +exec 3<&-; \ echo \"$child\" > \"$pid_file\"; \ wait \"$child\"; \ status=$?; \ @@ -804,6 +794,213 @@ exit \"$status\"\ ) } +fn docker_stdio_exec_options( + command: String, + working_dir: String, + env: Option>, +) -> (CreateExecOptions, StartExecOptions) { + ( + CreateExecOptions { + attach_stdin: Some(true), + attach_stdout: Some(true), + attach_stderr: Some(true), + tty: Some(false), + cmd: Some(vec!["/bin/bash".to_string(), "-lc".to_string(), command]), + working_dir: Some(working_dir), + env, + ..Default::default() + }, + StartExecOptions { + detach: false, + tty: false, + output_capacity: None, + }, + ) +} + +async fn create_and_start_exec( + docker: &Docker, + container_id: &str, + exec_options: CreateExecOptions, + start_options: Option, + create_context: &'static str, + start_context: &'static str, +) -> crate::Result<(String, StartExecResults)> { + let exec_instance = docker + .create_exec(container_id, exec_options) + .await + .map_err(|err| crate::Error::context(create_context, err))?; + let exec_id = exec_instance.id; + let start_result = docker + .start_exec(&exec_id, start_options) + .await + .map_err(|err| crate::Error::context(start_context, err))?; + + Ok((exec_id, start_result)) +} + +async fn request_docker_exec_stop_with( + docker: &Docker, + container_id: &str, + stop_file: &str, +) -> crate::Result<()> { + let command = format!("touch {}", shell_quote(stop_file)); + let exec_opts = CreateExecOptions { + cmd: Some(vec!["/bin/bash".to_string(), "-lc".to_string(), command]), + attach_stdout: Some(true), + attach_stderr: Some(true), + working_dir: Some("/".to_string()), + ..Default::default() + }; + let (exec_id, start_result) = create_and_start_exec( + docker, + container_id, + exec_opts, + None, + "Failed to create Docker exec stop request", + "Failed to start Docker exec stop request", + ) + .await?; + + let mut stdout = String::new(); + let mut stderr = String::new(); + if let StartExecResults::Attached { mut output, .. } = start_result { + while let Some(chunk) = output.next().await { + match chunk { + Ok(LogOutput::StdOut { message }) => { + stdout.push_str(&String::from_utf8_lossy(&message)); + } + Ok(LogOutput::StdErr { message }) => { + stderr.push_str(&String::from_utf8_lossy(&message)); + } + Ok(_) => {} + Err(e) => { + return Err(crate::Error::context( + "Error reading stop request output", + e, + )); + } + } + } + } + + let inspect = docker + .inspect_exec(&exec_id) + .await + .map_err(|e| crate::Error::context("Failed to inspect Docker exec stop request", e))?; + let exit_code = inspect + .exit_code + .and_then(|code| i32::try_from(code).ok()) + .unwrap_or(-1); + if exit_code != 0 { + return Err(crate::Error::message(format!( + "Failed to request Docker exec stop (exit {exit_code}): {stderr}{stdout}" + ))); + } + Ok(()) +} + +struct DockerStdioProcessControl { + docker: Docker, + container_id: String, + exec_id: String, + stop_file: String, + state: Arc, +} + +#[derive(Default)] +struct DockerStdioProcessState { + stop_requested: AtomicBool, + termination: TokioMutex>, + termination_notify: Notify, +} + +impl DockerStdioProcessState { + async fn cached_termination(&self) -> Option { + *self.termination.lock().await + } + + async fn request_stop_once(&self) -> bool { + self.cached_termination().await.is_none() + && !self.stop_requested.swap(true, Ordering::AcqRel) + } + + async fn cache_termination(&self, termination: StdioProcessTermination) { + let mut cached = self.termination.lock().await; + if cached.is_none() { + *cached = Some(termination); + self.termination_notify.notify_waiters(); + } + } + + async fn wait_for_cached_termination(&self) -> StdioProcessTermination { + loop { + if let Some(termination) = self.cached_termination().await { + return termination; + } + self.termination_notify.notified().await; + } + } +} + +#[async_trait] +impl StdioProcessControl for DockerStdioProcessControl { + async fn terminate(&self) -> crate::Result<()> { + if !self.state.request_stop_once().await { + return Ok(()); + } + request_docker_exec_stop_with(&self.docker, &self.container_id, &self.stop_file).await?; + Ok(()) + } + + async fn wait(&self) -> crate::Result { + if let Some(termination) = self.state.cached_termination().await { + return Ok(termination); + } + + let mut poll_interval = time::interval(std::time::Duration::from_secs(1)); + loop { + if let Some(termination) = self.state.cached_termination().await { + return Ok(termination); + } + let inspect = self + .docker + .inspect_exec(&self.exec_id) + .await + .map_err(|e| crate::Error::context("Failed to inspect Docker stdio exec", e))?; + if inspect.running != Some(true) { + let exit_code = inspect.exit_code.and_then(|code| i32::try_from(code).ok()); + let termination = StdioProcessTermination::exited(exit_code); + self.state.cache_termination(termination).await; + return Ok(termination); + } + tokio::select! { + termination = self.state.wait_for_cached_termination() => return Ok(termination), + _ = poll_interval.tick() => {} + } + } + } +} + +async fn cache_docker_stdio_completion( + docker: Docker, + exec_id: String, + state: Arc, +) { + match docker.inspect_exec(&exec_id).await { + Ok(inspect) if inspect.running != Some(true) => { + let exit_code = inspect.exit_code.and_then(|code| i32::try_from(code).ok()); + state + .cache_termination(StdioProcessTermination::exited(exit_code)) + .await; + } + Ok(_) => {} + Err(err) => { + tracing::warn!(error = %err, "Failed to inspect completed Docker stdio exec"); + } + } +} + fn git_clone_command(clone_url: &str, branch: Option<&str>) -> String { let mut command = "git -c maintenance.auto=0 -c gc.auto=0 clone".to_string(); if let Some(branch) = branch { @@ -1333,6 +1530,98 @@ impl Sandbox for DockerSandbox { .await } + async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&HashMap>, + cancel_token: Option, + ) -> crate::Result { + let effective_dir = working_dir.map_or_else( + || WORKING_DIRECTORY.to_string(), + Self::resolve_container_path, + ); + let env: Option> = + env_vars.map(|vars| vars.iter().map(|(k, v)| format!("{k}={v}")).collect()); + let (stop_file, pid_file) = docker_exec_control_paths(); + let controlled_command = docker_controlled_shell_command(command, &stop_file, &pid_file); + let (create_opts, start_opts) = + docker_stdio_exec_options(controlled_command, effective_dir, env); + + let container_id = self.container_id()?.to_string(); + let (exec_id, start_result) = create_and_start_exec( + &self.docker, + &container_id, + create_opts, + Some(start_opts), + "Failed to create Docker stdio exec", + "Failed to start Docker stdio exec", + ) + .await?; + + let StartExecResults::Attached { mut output, input } = start_result else { + return Err(crate::Error::message( + "Docker stdio exec started detached unexpectedly", + )); + }; + + let stderr_collector = StderrCollector::new(DEFAULT_EXEC_OUTPUT_TAIL_BYTES); + let stderr_for_output = stderr_collector.clone(); + let (mut stdout_writer, stdout_reader) = duplex(64 * 1024); + let state = Arc::new(DockerStdioProcessState::default()); + let state_for_output = Arc::clone(&state); + let docker_for_output = self.docker.clone(); + let exec_id_for_output = exec_id.clone(); + tokio::spawn(async move { + while let Some(chunk) = output.next().await { + match chunk { + Ok(LogOutput::StdOut { message }) => { + if let Err(err) = stdout_writer.write_all(&message).await { + tracing::warn!(error = %err, "Failed to forward Docker stdio stdout"); + break; + } + } + Ok(LogOutput::StdErr { message }) => { + stderr_for_output.push(&message).await; + } + Ok(_) => {} + Err(err) => { + let message = format!("Docker stdio output stream error: {err}"); + stderr_for_output.push(message.as_bytes()).await; + break; + } + } + } + cache_docker_stdio_completion(docker_for_output, exec_id_for_output, state_for_output) + .await; + }); + + let handle = StdioProcessHandle::new(DockerStdioProcessControl { + docker: self.docker.clone(), + container_id, + exec_id, + stop_file, + state, + }); + + if let Some(token) = cancel_token { + let handle_for_cancel = handle.clone(); + tokio::spawn(async move { + token.cancelled().await; + if let Err(err) = handle_for_cancel.terminate().await { + tracing::warn!(error = %err, "Failed to terminate cancelled Docker stdio exec"); + } + }); + } + + Ok(StdioProcess { + stdin: input, + stdout: Box::pin(stdout_reader), + stderr: stderr_collector, + handle, + }) + } + async fn read_file( &self, path: &str, @@ -1666,8 +1955,10 @@ mod tests { reason = "unit test reads an in-memory tar entry synchronously" )] use std::io::Read as _; + use std::process::Stdio; use std::time::Duration; + use tokio::io::AsyncWriteExt as _; use tokio::process::Command; use super::*; @@ -1744,6 +2035,33 @@ mod tests { ); } + #[test] + fn stdio_exec_options_attach_streams_without_tty() { + let (create, start) = docker_stdio_exec_options( + "python fake_agent.py".to_string(), + WORKING_DIRECTORY.to_string(), + Some(vec!["MODE=test".to_string()]), + ); + + assert_eq!(create.attach_stdin, Some(true)); + assert_eq!(create.attach_stdout, Some(true)); + assert_eq!(create.attach_stderr, Some(true)); + assert_eq!(create.tty, Some(false)); + assert_eq!(create.working_dir.as_deref(), Some(WORKING_DIRECTORY)); + assert_eq!(create.env, Some(vec!["MODE=test".to_string()])); + assert_eq!( + create.cmd, + Some(vec![ + "/bin/bash".to_string(), + "-lc".to_string(), + "python fake_agent.py".to_string() + ]) + ); + assert!(!start.detach); + assert!(!start.tty); + assert_eq!(start.output_capacity, None); + } + #[tokio::test] async fn controlled_shell_command_honors_stop_requested_before_pid_file_exists() { let tempdir = tempfile::tempdir().expect("tempdir should be created"); @@ -1788,6 +2106,59 @@ mod tests { ); } + #[tokio::test] + async fn controlled_shell_command_preserves_stdin_for_user_command() { + let tempdir = tempfile::tempdir().expect("tempdir should be created"); + let stop_file = tempdir.path().join("stop"); + let pid_file = tempdir.path().join("pid"); + let stop_file = stop_file.to_string_lossy().into_owned(); + let pid_file = pid_file.to_string_lossy().into_owned(); + let command = docker_controlled_shell_command("cat", &stop_file, &pid_file); + + let mut child = Command::new("/bin/bash") + .arg("-lc") + .arg(command) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .kill_on_drop(true) + .spawn() + .expect("controlled shell command should spawn"); + let mut stdin = child + .stdin + .take() + .expect("controlled shell command stdin should be piped"); + stdin + .write_all(b"abc\n") + .await + .expect("stdin should be written"); + drop(stdin); + + let output = time::timeout(Duration::from_secs(5), child.wait_with_output()) + .await + .expect("controlled shell command should not hang") + .expect("controlled shell command should run"); + + assert!( + output.status.success(), + "controlled shell command should exit successfully: {output:?}" + ); + assert_eq!(output.stdout, b"abc\n"); + } + + #[tokio::test] + async fn docker_stdio_process_state_does_not_cache_cancelled_on_stop_request() { + let state = DockerStdioProcessState::default(); + + assert!(state.request_stop_once().await); + assert_eq!(state.cached_termination().await, None); + assert!(!state.request_stop_once().await); + + let termination = StdioProcessTermination::exited(Some(143)); + state.cache_termination(termination).await; + assert_eq!(state.cached_termination().await, Some(termination)); + assert!(!state.request_stop_once().await); + } + #[tokio::test] async fn controlled_shell_command_skips_user_command_when_stop_already_requested() { let tempdir = tempfile::tempdir().expect("tempdir should be created"); diff --git a/lib/crates/fabro-sandbox/src/lib.rs b/lib/crates/fabro-sandbox/src/lib.rs index a0902a46f..04ca2fb86 100644 --- a/lib/crates/fabro-sandbox/src/lib.rs +++ b/lib/crates/fabro-sandbox/src/lib.rs @@ -41,7 +41,8 @@ pub use reconnect::{reconnect, reconnect_for_run, reconnect_for_run_with_callbac pub use sandbox::{ CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult, ExecStreamingResult, GitRunInfo, GitSetupIntent, GrepOptions, Sandbox, SandboxEvent, - SandboxEventCallback, format_lines_numbered, git_push_via_exec, redacted_output_tail, + SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle, + StdioProcessTermination, format_lines_numbered, git_push_via_exec, redacted_output_tail, setup_git_via_exec, shell_quote, }; pub use sandbox_spec::SandboxSpec; diff --git a/lib/crates/fabro-sandbox/src/local.rs b/lib/crates/fabro-sandbox/src/local.rs index 9b5091a61..dc94c35f3 100644 --- a/lib/crates/fabro-sandbox/src/local.rs +++ b/lib/crates/fabro-sandbox/src/local.rs @@ -7,14 +7,16 @@ use fabro_types::{CommandOutputStream, CommandTermination}; use fabro_util::time::elapsed_ms; use tokio::io::{AsyncRead, AsyncReadExt}; use tokio::process::{Child, Command}; +use tokio::sync::watch; use tokio::task::spawn_blocking; use tokio::{fs, time}; use tokio_util::sync::CancellationToken; -use crate::sandbox::optional_timeout; +use crate::sandbox::{StdioProcessControl, optional_timeout}; use crate::{ - CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GrepOptions, Sandbox, - SandboxEvent, SandboxEventCallback, format_lines_numbered, + CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult, + ExecStreamingResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback, StderrCollector, + StdioProcess, StdioProcessHandle, StdioProcessTermination, format_lines_numbered, }; pub struct LocalSandbox { @@ -128,6 +130,34 @@ fn process_env_vars() -> Vec<(String, String)> { std::env::vars().collect() } +#[derive(Debug, Clone, Copy)] +enum ExplicitEnvPolicy { + FilterSensitive, + TrustCaller, +} + +fn filtered_env_vars( + env_vars: Option<&std::collections::HashMap>, + explicit_policy: ExplicitEnvPolicy, +) -> Vec<(String, String)> { + let mut filtered_env: Vec<(String, String)> = process_env_vars() + .into_iter() + .filter(|(key, _)| !LocalSandbox::should_filter_env_var(key)) + .collect(); + + if let Some(extra) = env_vars { + for (key, value) in extra { + if matches!(explicit_policy, ExplicitEnvPolicy::TrustCaller) + || !LocalSandbox::should_filter_env_var(key) + { + filtered_env.push((key.clone(), value.clone())); + } + } + } + + filtered_env +} + async fn drain_pipe(mut pipe: Option, stream: CommandOutputStream) -> String where R: AsyncRead + Unpin, @@ -141,6 +171,77 @@ where buf } +type LocalStdioOutcome = Result; + +struct LocalStdioProcessControl { + terminate_tx: watch::Sender, + termination_rx: watch::Receiver>, +} + +impl LocalStdioProcessControl { + fn new(mut child: Child) -> Self { + let (terminate_tx, mut terminate_rx) = watch::channel(false); + let (termination_tx, termination_rx) = watch::channel(None); + + tokio::spawn(async move { + let outcome = tokio::select! { + status = child.wait() => { + status + .map(|status| StdioProcessTermination::exited(status.code())) + .map_err(|err| format!("Failed to wait for stdio process: {err}")) + } + changed = terminate_rx.changed() => { + if changed.is_err() || !*terminate_rx.borrow() { + child.wait() + .await + .map(|status| StdioProcessTermination::exited(status.code())) + .map_err(|err| format!("Failed to wait for stdio process: {err}")) + } else { + sigterm_then_kill(&mut child).await; + Ok(StdioProcessTermination::cancelled()) + } + } + }; + let _ = termination_tx.send(Some(outcome)); + }); + + Self { + terminate_tx, + termination_rx, + } + } + + async fn wait_for_termination(&self) -> crate::Result { + let mut termination_rx = self.termination_rx.clone(); + loop { + if let Some(outcome) = termination_rx.borrow().clone() { + return outcome.map_err(crate::Error::message); + } + termination_rx.changed().await.map_err(|_| { + crate::Error::message( + "stdio process supervisor stopped before reporting termination", + ) + })?; + } + } +} + +#[async_trait] +impl StdioProcessControl for LocalStdioProcessControl { + async fn terminate(&self) -> crate::Result<()> { + if self.termination_rx.borrow().is_some() { + return Ok(()); + } + + self.terminate_tx.send_replace(true); + self.wait_for_termination().await.map(|_| ()) + } + + async fn wait(&self) -> crate::Result { + self.wait_for_termination().await + } +} + #[async_trait] impl Sandbox for LocalSandbox { async fn read_file( @@ -252,18 +353,7 @@ impl Sandbox for LocalSandbox { ) -> crate::Result { let start = Instant::now(); - let mut filtered_env: Vec<(String, String)> = process_env_vars() - .into_iter() - .filter(|(key, _)| !Self::should_filter_env_var(key)) - .collect(); - - if let Some(extra) = env_vars { - for (k, v) in extra { - if !Self::should_filter_env_var(k) { - filtered_env.push((k.clone(), v.clone())); - } - } - } + let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::FilterSensitive); let effective_dir = working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from); @@ -340,18 +430,7 @@ impl Sandbox for LocalSandbox { ) -> crate::Result { let start = Instant::now(); - let mut filtered_env: Vec<(String, String)> = process_env_vars() - .into_iter() - .filter(|(key, _)| !Self::should_filter_env_var(key)) - .collect(); - - if let Some(extra) = env_vars { - for (k, v) in extra { - if !Self::should_filter_env_var(k) { - filtered_env.push((k.clone(), v.clone())); - } - } - } + let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::FilterSensitive); let effective_dir = working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from); @@ -424,6 +503,71 @@ impl Sandbox for LocalSandbox { }) } + async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&std::collections::HashMap>, + cancel_token: Option, + ) -> crate::Result { + let filtered_env = filtered_env_vars(env_vars, ExplicitEnvPolicy::TrustCaller); + + let effective_dir = + working_dir.map_or_else(|| self.working_directory.clone(), std::path::PathBuf::from); + + let mut cmd = Command::new("/bin/bash"); + cmd.arg("-lc") + .arg(format!("exec {command}")) + .current_dir(&effective_dir) + .env_clear() + .envs(filtered_env) + .stdin(std::process::Stdio::piped()) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()); + + #[cfg(unix)] + fabro_proc::pre_exec_setpgid(cmd.as_std_mut()); + + let mut child = cmd + .spawn() + .map_err(|e| crate::Error::context("Failed to spawn stdio process", e))?; + + let stdin = child + .stdin + .take() + .ok_or_else(|| crate::Error::message("Failed to open stdio process stdin"))?; + let stdout = child + .stdout + .take() + .ok_or_else(|| crate::Error::message("Failed to open stdio process stdout"))?; + let stderr = child + .stderr + .take() + .ok_or_else(|| crate::Error::message("Failed to open stdio process stderr"))?; + + let stderr_collector = StderrCollector::new(DEFAULT_EXEC_OUTPUT_TAIL_BYTES); + stderr_collector.spawn_reader(stderr); + + let handle = StdioProcessHandle::new(LocalStdioProcessControl::new(child)); + + if let Some(token) = cancel_token { + let handle_for_cancel = handle.clone(); + tokio::spawn(async move { + token.cancelled().await; + if let Err(err) = handle_for_cancel.terminate().await { + tracing::warn!(error = %err, "Failed to terminate cancelled stdio process"); + } + }); + } + + Ok(StdioProcess { + stdin: Box::pin(stdin), + stdout: Box::pin(stdout), + stderr: stderr_collector, + handle, + }) + } + async fn grep( &self, pattern: &str, @@ -740,7 +884,7 @@ mod tests { use std::pin::Pin; use std::task::{Context as TaskContext, Poll}; - use tokio::io::ReadBuf; + use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader, ReadBuf}; use super::*; @@ -876,6 +1020,58 @@ mod tests { std::fs::remove_dir_all(&dir).unwrap(); } + #[tokio::test] + async fn stdio_process_round_trips_lines() { + let dir = temp_dir(); + let sandbox = LocalSandbox::new(dir.clone()); + let process = sandbox + .spawn_stdio_process( + "python3 -u -c 'import sys; [print(line.strip()[::-1], flush=True) for line in sys.stdin]'", + None, + None, + None, + ) + .await + .unwrap(); + + let mut stdin = process.stdin; + let mut stdout = BufReader::new(process.stdout); + + stdin.write_all(b"abc\n").await.unwrap(); + stdin.flush().await.unwrap(); + + let mut line = String::new(); + stdout.read_line(&mut line).await.unwrap(); + assert_eq!(line.trim_end(), "cba"); + + process.handle.terminate().await.unwrap(); + std::fs::remove_dir_all(&dir).unwrap(); + } + + #[tokio::test] + async fn stdio_process_forwards_explicit_provider_credentials() { + let dir = temp_dir(); + let sandbox = LocalSandbox::new(dir.clone()); + let env = HashMap::from([("OPENAI_API_KEY".to_string(), "test-key".to_string())]); + let process = sandbox + .spawn_stdio_process( + "python3 -u -c 'import os; print(os.environ.get(\"OPENAI_API_KEY\", \"missing\"), flush=True)'", + None, + Some(&env), + None, + ) + .await + .unwrap(); + + let mut stdout = BufReader::new(process.stdout); + let mut line = String::new(); + stdout.read_line(&mut line).await.unwrap(); + assert_eq!(line.trim_end(), "test-key"); + + process.handle.wait().await.unwrap(); + std::fs::remove_dir_all(&dir).unwrap(); + } + #[tokio::test] async fn exec_command_exit_code() { let dir = temp_dir(); diff --git a/lib/crates/fabro-sandbox/src/read_guard.rs b/lib/crates/fabro-sandbox/src/read_guard.rs index 74b6d0c29..588bd960d 100644 --- a/lib/crates/fabro-sandbox/src/read_guard.rs +++ b/lib/crates/fabro-sandbox/src/read_guard.rs @@ -277,4 +277,22 @@ mod tests { assert!(result.is_ok()); } + + #[tokio::test] + async fn stdio_process_forwards_to_inner_sandbox() { + let mock = Arc::new(MockSandbox::linux()); + let env = ReadBeforeWriteSandbox::new(mock.clone()); + + env.spawn_stdio_process("python fake_agent.py", Some("/work/sub"), None, None) + .await + .unwrap(); + + assert_eq!( + *mock.captured_command.lock().unwrap(), + Some("python fake_agent.py".to_string()) + ); + assert_eq!(*mock.captured_working_dirs.lock().unwrap(), vec![Some( + "/work/sub".to_string() + )]); + } } diff --git a/lib/crates/fabro-sandbox/src/sandbox.rs b/lib/crates/fabro-sandbox/src/sandbox.rs index ef7ae3b90..4679f972a 100644 --- a/lib/crates/fabro-sandbox/src/sandbox.rs +++ b/lib/crates/fabro-sandbox/src/sandbox.rs @@ -9,6 +9,9 @@ use std::time::Duration; use async_trait::async_trait; use fabro_types::{CommandOutputStream, CommandTermination}; use serde::{Deserialize, Serialize}; +use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite}; +use tokio::sync::Mutex as TokioMutex; +use tokio::task::JoinHandle; use tokio::time; use tokio_util::sync::CancellationToken; @@ -121,6 +124,18 @@ macro_rules! delegate_sandbox { .await } + async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&std::collections::HashMap>, + cancel_token: Option, + ) -> $crate::Result<$crate::StdioProcess> { + self.$field + .spawn_stdio_process(command, working_dir, env_vars, cancel_token) + .await + } + async fn glob(&self, pattern: &str, path: Option<&str>) -> $crate::Result> { self.$field.glob(pattern, path).await } @@ -666,6 +681,114 @@ pub type CommandOutputCallback = Arc< + Sync, >; +pub struct StdioProcess { + pub stdin: Pin>, + pub stdout: Pin>, + pub stderr: StderrCollector, + pub handle: StdioProcessHandle, +} + +#[derive(Debug, Clone)] +pub struct StderrCollector { + inner: Arc>>, + max_bytes: usize, +} + +impl StderrCollector { + #[must_use] + pub fn new(max_bytes: usize) -> Self { + Self { + inner: Arc::new(TokioMutex::new(Vec::new())), + max_bytes, + } + } + + pub async fn push(&self, bytes: &[u8]) { + let mut tail = self.inner.lock().await; + tail.extend_from_slice(bytes); + if tail.len() > self.max_bytes { + let excess = tail.len() - self.max_bytes; + tail.drain(..excess); + } + } + + pub async fn tail_string(&self) -> String { + let tail = self.inner.lock().await; + String::from_utf8_lossy(&tail).into_owned() + } + + pub fn spawn_reader(&self, mut reader: R) -> JoinHandle<()> + where + R: AsyncRead + Unpin + Send + 'static, + { + let collector = self.clone(); + tokio::spawn(async move { + let mut buf = [0_u8; 8192]; + loop { + match reader.read(&mut buf).await { + Ok(0) => return, + Ok(read) => collector.push(&buf[..read]).await, + Err(err) => { + tracing::warn!(error = %err, "Failed to read stdio process stderr"); + return; + } + } + } + }) + } +} + +#[derive(Clone)] +pub struct StdioProcessHandle { + control: Arc, +} + +impl StdioProcessHandle { + pub(crate) fn new(control: impl StdioProcessControl + 'static) -> Self { + Self { + control: Arc::new(control), + } + } + + pub async fn terminate(&self) -> crate::Result<()> { + self.control.terminate().await + } + + pub async fn wait(&self) -> crate::Result { + self.control.wait().await + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct StdioProcessTermination { + pub termination: CommandTermination, + pub exit_code: Option, +} + +impl StdioProcessTermination { + #[must_use] + pub fn exited(exit_code: Option) -> Self { + Self { + termination: CommandTermination::Exited, + exit_code, + } + } + + #[must_use] + pub fn cancelled() -> Self { + Self { + termination: CommandTermination::Cancelled, + exit_code: None, + } + } +} + +#[async_trait] +pub(crate) trait StdioProcessControl: Send + Sync { + async fn terminate(&self) -> crate::Result<()>; + async fn wait(&self) -> crate::Result; +} + #[derive(Debug, Clone)] pub struct DirEntry { pub name: String, @@ -752,6 +875,19 @@ pub trait Sandbox: Send + Sync { live_streaming: false, }) } + + async fn spawn_stdio_process( + &self, + _command: &str, + _working_dir: Option<&str>, + _env_vars: Option<&HashMap>, + _cancel_token: Option, + ) -> crate::Result { + Err(crate::Error::message( + "ACP backend requires bidirectional stdio; this sandbox provider does not support it", + )) + } + async fn grep( &self, pattern: &str, diff --git a/lib/crates/fabro-sandbox/src/test_support.rs b/lib/crates/fabro-sandbox/src/test_support.rs index 0c7af3dd9..42cf769d5 100644 --- a/lib/crates/fabro-sandbox/src/test_support.rs +++ b/lib/crates/fabro-sandbox/src/test_support.rs @@ -4,9 +4,15 @@ use std::sync::Mutex; use async_trait::async_trait; use fabro_types::CommandTermination; use tokio::fs; +use tokio::io::duplex; use tokio_util::sync::CancellationToken; -use crate::{DirEntry, ExecResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback}; +use crate::sandbox::StdioProcessControl; +use crate::{ + DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult, GrepOptions, Sandbox, SandboxEvent, + SandboxEventCallback, StderrCollector, StdioProcess, StdioProcessHandle, + StdioProcessTermination, +}; // --- MockSandbox --- @@ -36,6 +42,7 @@ pub struct MockSandbox { pub stop_calls: Mutex, pub delete_calls: Mutex, pub event_callback: Option, + pub stdio_process_error: Option, } impl MockSandbox { @@ -100,10 +107,24 @@ impl Default for MockSandbox { stop_calls: Mutex::new(0), delete_calls: Mutex::new(0), event_callback: None, + stdio_process_error: None, } } } +struct MockStdioProcessControl; + +#[async_trait] +impl StdioProcessControl for MockStdioProcessControl { + async fn terminate(&self) -> crate::Result<()> { + Ok(()) + } + + async fn wait(&self) -> crate::Result { + Ok(StdioProcessTermination::exited(Some(0))) + } +} + #[async_trait] impl Sandbox for MockSandbox { async fn read_file( @@ -184,6 +205,44 @@ impl Sandbox for MockSandbox { Ok(self.exec_result.clone()) } + async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&std::collections::HashMap>, + _cancel_token: Option, + ) -> crate::Result { + *self + .captured_command + .lock() + .expect("captured_command lock poisoned") = Some(command.to_string()); + self.captured_commands + .lock() + .expect("captured_commands lock poisoned") + .push(command.to_string()); + self.captured_working_dirs + .lock() + .expect("captured_working_dirs lock poisoned") + .push(working_dir.map(String::from)); + *self + .captured_env_vars + .lock() + .expect("captured_env_vars lock poisoned") = env_vars.cloned(); + + if let Some(error) = &self.stdio_process_error { + return Err(crate::Error::message(error.clone())); + } + + let (stdin, _stdin_read) = duplex(1024); + let (_stdout_write, stdout) = duplex(1024); + Ok(StdioProcess { + stdin: Box::pin(stdin), + stdout: Box::pin(stdout), + stderr: StderrCollector::new(DEFAULT_EXEC_OUTPUT_TAIL_BYTES), + handle: StdioProcessHandle::new(MockStdioProcessControl), + }) + } + async fn grep( &self, _pattern: &str, diff --git a/lib/crates/fabro-sandbox/src/worktree.rs b/lib/crates/fabro-sandbox/src/worktree.rs index 87a31db31..39ef168c7 100644 --- a/lib/crates/fabro-sandbox/src/worktree.rs +++ b/lib/crates/fabro-sandbox/src/worktree.rs @@ -8,7 +8,7 @@ use tokio_util::sync::CancellationToken; use crate::sandbox::fetch_source_run_ref; use crate::{ CommandOutputCallback, DirEntry, ExecResult, ExecStreamingResult, GitRunInfo, GitSetupIntent, - GrepOptions, Sandbox, shell_quote, + GrepOptions, Sandbox, StdioProcess, shell_quote, }; /// Git command prefix that disables background maintenance. @@ -267,6 +267,19 @@ impl Sandbox for WorktreeSandbox { .await } + async fn spawn_stdio_process( + &self, + command: &str, + working_dir: Option<&str>, + env_vars: Option<&HashMap>, + cancel_token: Option, + ) -> crate::Result { + let wd = working_dir.unwrap_or(&self.config.worktree_path); + self.inner + .spawn_stdio_process(command, Some(wd), env_vars, cancel_token) + .await + } + // --- Delegated methods --- async fn read_file( @@ -676,6 +689,23 @@ mod tests { ); } + #[tokio::test] + async fn stdio_process_none_working_dir_defaults_to_worktree_path() { + let (inner, mock) = make_mock(); + let wt = WorktreeSandbox::new(inner, make_config("/tmp/wt")); + + wt.spawn_stdio_process("python fake_agent.py", None, None, None) + .await + .unwrap(); + + let wdirs = mock.captured_working_dirs.lock().unwrap().clone(); + assert_eq!( + wdirs.last(), + Some(&Some("/tmp/wt".to_string())), + "None working_dir should be replaced with worktree path" + ); + } + // ----------------------------------------------------------------------- // Accessors // ----------------------------------------------------------------------- diff --git a/lib/crates/fabro-server/Cargo.toml b/lib/crates/fabro-server/Cargo.toml index e78933660..304ae45ec 100644 --- a/lib/crates/fabro-server/Cargo.toml +++ b/lib/crates/fabro-server/Cargo.toml @@ -35,6 +35,7 @@ fabro-sandbox = { path = "../fabro-sandbox", features = ["daytona", "docker"] } fabro-github = { path = "../fabro-github" } fabro-agent = { path = "../fabro-agent" } fabro-llm = { path = "../fabro-llm" } +fabro-manifest = { path = "../fabro-manifest" } fabro-model = { path = "../fabro-model" } fabro-proc = { path = "../fabro-proc" } fabro-types = { path = "../fabro-types" } diff --git a/lib/crates/fabro-server/src/run_manifest.rs b/lib/crates/fabro-server/src/run_manifest.rs index 6de1c2b0d..2178196ee 100644 --- a/lib/crates/fabro-server/src/run_manifest.rs +++ b/lib/crates/fabro-server/src/run_manifest.rs @@ -9,8 +9,7 @@ use fabro_api::types; use fabro_auth::auth_issue_message; use fabro_config::run::parse_run_layer_from_settings_toml; use fabro_config::{ - CliLayer, CliOutputLayer, DaytonaDockerfileLayer, DockerSandboxLayer, ReplaceMap, - RunExecutionLayer, RunLayer, RunModelLayer, RunSandboxLayer, WorkflowSettingsBuilder, + CliLayer, CliOutputLayer, DaytonaDockerfileLayer, RunLayer, WorkflowSettingsBuilder, parse_input_overrides, }; use fabro_graphviz::graph::{Graph, is_llm_handler_type}; @@ -28,8 +27,8 @@ use fabro_static::EnvVars; use fabro_types::settings::cli::OutputVerbosity; use fabro_types::settings::interp::InterpString; use fabro_types::settings::run::{ - ApprovalMode, DaytonaNetworkLayer, DaytonaSettings, DockerSettings, DockerfileSource, RunGoal, - RunMode, RunNamespace, + DaytonaNetworkLayer, DaytonaSettings, DockerSettings, DockerfileSource, RunGoal, RunMode, + RunNamespace, }; use fabro_types::{RunId, WorkflowSettings}; use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSection, CheckStatus}; @@ -316,47 +315,16 @@ fn manifest_args_overrides( return Ok(ManifestSettingsOverrides::default()); }; - let model = (args.model.is_some() || args.provider.is_some()).then(|| RunModelLayer { - provider: args.provider.as_deref().map(InterpString::parse), - name: args.model.as_deref().map(InterpString::parse), - fallbacks: Vec::new(), - controls: None, - }); - let sandbox = - (args.sandbox.is_some() || args.preserve_sandbox.is_some() || args.docker_image.is_some()) - .then(|| RunSandboxLayer { - provider: args.sandbox.clone(), - preserve: args.preserve_sandbox, - docker: args.docker_image.as_ref().map(|image| DockerSandboxLayer { - image: Some(image.clone()), - ..DockerSandboxLayer::default() - }), - ..RunSandboxLayer::default() - }); - - let execution_has_any = args.dry_run.is_some() || args.auto_approve.is_some(); - let execution = execution_has_any.then(|| RunExecutionLayer { - mode: args - .dry_run - .map(|d| if d { RunMode::DryRun } else { RunMode::Normal }), - approval: args.auto_approve.map(|a| { - if a { - ApprovalMode::Auto - } else { - ApprovalMode::Prompt - } - }), - }); - - let run_has_any = - model.is_some() || sandbox.is_some() || execution.is_some() || !args.label.is_empty(); - - let run = run_has_any.then(|| RunLayer { - model, - sandbox, - execution, - metadata: ReplaceMap::from(parse_labels(&args.label)), - ..RunLayer::default() + let run = fabro_manifest::build_sparse_run_overrides(fabro_manifest::RunOverrideInput { + goal: None, + model: args.model.as_deref(), + provider: args.provider.as_deref(), + sandbox: args.sandbox.as_deref(), + docker_image: args.docker_image.as_deref(), + preserve_sandbox: args.preserve_sandbox, + dry_run: args.dry_run, + auto_approve: args.auto_approve, + labels: parse_labels(&args.label), }); // Verbose is a CLI output concern in v2; route it through cli.output.verbosity. diff --git a/lib/crates/fabro-server/src/server.rs b/lib/crates/fabro-server/src/server.rs index e099f8d75..1cefd3507 100644 --- a/lib/crates/fabro-server/src/server.rs +++ b/lib/crates/fabro-server/src/server.rs @@ -189,30 +189,30 @@ impl ListResponse { /// Snapshot of a managed run. struct ManagedRun { - dot_source: String, - status: RunStatus, - error: Option, - created_at: chrono::DateTime, - enqueued_at: Instant, + dot_source: String, + status: RunStatus, + error: Option, + created_at: chrono::DateTime, + enqueued_at: Instant, // Populated when running: - answer_transport: Option, + answer_transport: Option, accepted_questions: HashSet, /// Stage IDs of currently steerable API-mode (SDK) agent sessions, /// keyed to the session id that owns the active lease. Used by the /// steerability predicate. - active_api_stages: HashMap, - /// Stage IDs of currently running CLI-mode agent sessions, observed - /// from `agent.cli.started/completed` plus `stage.completed`/ + active_api_stages: HashMap, + /// Stage IDs of currently running non-steerable agent sessions, observed + /// from CLI/ACP start/completion events plus `stage.completed`/ /// `stage.failed` backstops. - active_cli_stages: HashSet, - event_tx: Option>, - checkpoint: Option, - cancel_tx: Option>, - cancel_token: Option, - worker_pid: Option, - worker_pgid: Option, - run_dir: Option, - execution_mode: RunExecutionMode, + active_non_steerable_agent_stages: HashSet, + event_tx: Option>, + checkpoint: Option, + cancel_tx: Option>, + cancel_token: Option, + worker_pid: Option, + worker_pgid: Option, + run_dir: Option, + execution_mode: RunExecutionMode, } #[derive(Clone, Copy)] @@ -1953,7 +1953,7 @@ fn clear_live_run_state(run: &mut ManagedRun) { run.answer_transport = None; run.accepted_questions.clear(); run.active_api_stages.clear(); - run.active_cli_stages.clear(); + run.active_non_steerable_agent_stages.clear(); run.event_tx = None; run.cancel_tx = None; run.cancel_token = None; @@ -2291,7 +2291,7 @@ fn managed_run( answer_transport: None, accepted_questions: HashSet::new(), active_api_stages: HashMap::new(), - active_cli_stages: HashSet::new(), + active_non_steerable_agent_stages: HashSet::new(), event_tx: None, checkpoint: None, cancel_tx: None, @@ -2322,6 +2322,15 @@ async fn load_pending_control( .and_then(|summary| summary.lifecycle.pending_control)) } +async fn durable_run_status(state: &AppState, run_id: RunId) -> anyhow::Result> { + Ok(state + .store + .runs() + .find(&run_id) + .await? + .map(|summary| summary.lifecycle.status)) +} + fn fail_managed_run(state: &Arc, run_id: RunId, reason: FailureReason, message: String) { let mut runs = state.runs.lock().expect("runs lock poisoned"); if let Some(managed_run) = runs.get_mut(&run_id) { @@ -2388,7 +2397,7 @@ fn update_live_run_from_event(state: &AppState, run_id: RunId, event: &RunEvent) }; managed_run.error = None; managed_run.active_api_stages.clear(); - managed_run.active_cli_stages.clear(); + managed_run.active_non_steerable_agent_stages.clear(); } EventBody::RunFailed(props) => { managed_run.status = RunStatus::Failed { @@ -2396,7 +2405,7 @@ fn update_live_run_from_event(state: &AppState, run_id: RunId, event: &RunEvent) }; managed_run.error = Some(props.error.clone()); managed_run.active_api_stages.clear(); - managed_run.active_cli_stages.clear(); + managed_run.active_non_steerable_agent_stages.clear(); } // Track API-mode steerable sessions. Activated/deactivated are // leased by session id so stale deactivations cannot clear a newer @@ -2425,17 +2434,24 @@ fn update_live_run_from_event(state: &AppState, run_id: RunId, event: &RunEvent) } } } - // Track CLI-mode agent stages. CLI started/completed are coarser - // and sometimes fail to emit `completed` on error paths — the - // stage.completed/stage.failed handler below is the backstop. - EventBody::AgentCliStarted(_) => { + // Track non-steerable agent stages. CLI/ACP started/completed are + // coarser and sometimes fail to emit terminal events on error paths; + // stage.completed/stage.failed below are the backstops. + EventBody::AgentCliStarted(_) | EventBody::AgentAcpStarted(_) => { if let Some(stage_id) = event.stage_id.as_ref() { - managed_run.active_cli_stages.insert(stage_id.clone()); + managed_run + .active_non_steerable_agent_stages + .insert(stage_id.clone()); } } - EventBody::AgentCliCompleted(_) => { + EventBody::AgentCliCompleted(_) + | EventBody::AgentAcpCompleted(_) + | EventBody::AgentAcpCancelled(_) + | EventBody::AgentAcpTimedOut(_) => { if let Some(stage_id) = &event.stage_id { - managed_run.active_cli_stages.remove(stage_id); + managed_run + .active_non_steerable_agent_stages + .remove(stage_id); } } // Stage lifecycle backstop: cover both completion and failure @@ -2443,7 +2459,9 @@ fn update_live_run_from_event(state: &AppState, run_id: RunId, event: &RunEvent) EventBody::StageCompleted(_) | EventBody::StageFailed(_) => { if let Some(stage_id) = &event.stage_id { managed_run.active_api_stages.remove(stage_id); - managed_run.active_cli_stages.remove(stage_id); + managed_run + .active_non_steerable_agent_stages + .remove(stage_id); } } _ => {} diff --git a/lib/crates/fabro-server/src/server/handler/lifecycle.rs b/lib/crates/fabro-server/src/server/handler/lifecycle.rs index 469f74bc1..80e5d2918 100644 --- a/lib/crates/fabro-server/src/server/handler/lifecycle.rs +++ b/lib/crates/fabro-server/src/server/handler/lifecycle.rs @@ -5,7 +5,7 @@ use super::super::{ Principal, RequiredUser, Response, RewindRequest, RewindResponse, Router, RunAnswerTransport, RunControlAction, RunExecutionMode, RunId, RunStatus, StartRunRequest, State, StatusCode, Storage, TimelineEntryResponse, WORKER_CANCEL_GRACE, WorkflowError, append_control_request, - get, load_pending_control, managed_run, operations, parse_run_id_path, + durable_run_status, get, load_pending_control, managed_run, operations, parse_run_id_path, persist_cancelled_run_status, post, reject_if_archived, sleep, update_live_run_from_event, workflow_event, }; @@ -171,7 +171,7 @@ async fn cancel_run( .into_response(); } }; - let (persist_cancelled_status, answer_transport, cancel_token, cancel_tx, worker_pid) = { + let cancel_target = { let mut runs = state.runs.lock().expect("runs lock poisoned"); match runs.get_mut(&id) { Some(managed_run) => match managed_run.status { @@ -192,7 +192,7 @@ async fn cancel_run( reason: FailureReason::Cancelled, }; } - ( + Some(( persist_cancelled_status, managed_run.answer_transport.clone(), managed_run.cancel_token.clone(), @@ -200,16 +200,21 @@ async fn cancel_run( .then(|| managed_run.cancel_tx.take()) .flatten(), managed_run.worker_pid, - ) + )) } _ => { return ApiError::new(StatusCode::CONFLICT, "Run is not cancellable.") .into_response(); } }, - None => return ApiError::not_found("Run not found.").into_response(), + None => None, } }; + let Some((persist_cancelled_status, answer_transport, cancel_token, cancel_tx, worker_pid)) = + cancel_target + else { + return unmanaged_cancel_response(state.as_ref(), id).await; + }; if pending_control != Some(RunControlAction::Cancel) { if let Err(err) = append_control_request( @@ -256,6 +261,23 @@ async fn cancel_run( run_response(state.as_ref(), id, StatusCode::OK).await } +async fn unmanaged_cancel_response(state: &AppState, id: RunId) -> Response { + match durable_run_status(state, id).await { + Ok(Some(status)) if status.is_terminal() => ApiError::new( + StatusCode::CONFLICT, + "Run is already terminal and cannot be cancelled.", + ) + .into_response(), + Ok(Some(_)) => { + ApiError::new(StatusCode::CONFLICT, "Run is not cancellable.").into_response() + } + Ok(None) => ApiError::not_found("Run not found.").into_response(), + Err(err) => { + ApiError::new(StatusCode::INTERNAL_SERVER_ERROR, err.to_string()).into_response() + } + } +} + /// How `pause_run` should enact the transition, chosen from the current run /// status. enum PauseMode { diff --git a/lib/crates/fabro-server/src/server/handler/steer.rs b/lib/crates/fabro-server/src/server/handler/steer.rs index 6ee20497c..069f37a4c 100644 --- a/lib/crates/fabro-server/src/server/handler/steer.rs +++ b/lib/crates/fabro-server/src/server/handler/steer.rs @@ -9,7 +9,9 @@ use fabro_api::types::SteerRunRequest; use fabro_types::Principal; use fabro_workflow::run_status::RunStatus; -use super::super::{AnswerTransportError, AppState, parse_run_id_path, reject_if_archived}; +use super::super::{ + AnswerTransportError, AppState, durable_run_status, parse_run_id_path, reject_if_archived, +}; use crate::error::ApiError; use crate::principal_middleware::RequiredUser; @@ -77,73 +79,72 @@ async fn control_run( // Status + steerability gate. Take the answer_transport snapshot under // the same lock so we can hand it off without further state races. - let answer_transport = { + let managed_answer_transport = { let runs = state.runs.lock().expect("runs lock poisoned"); - let Some(managed_run) = runs.get(&id) else { - return ApiError::not_found("Run not found.").into_response(); - }; - match managed_run.status { - RunStatus::Blocked { .. } => { - return ApiError::with_code( - StatusCode::CONFLICT, - "Run is blocked on a question; use the interview-answer endpoint instead.", - "use_answer_endpoint", - ) - .into_response(); + match runs.get(&id) { + Some(managed_run) => { + match managed_run.status { + RunStatus::Blocked { .. } => { + return ApiError::with_code( + StatusCode::CONFLICT, + "Run is blocked on a question; use the interview-answer endpoint \ + instead.", + "use_answer_endpoint", + ) + .into_response(); + } + RunStatus::Submitted + | RunStatus::Queued + | RunStatus::Starting + | RunStatus::Paused { .. } => { + return ApiError::with_code( + StatusCode::CONFLICT, + "Run is not currently running.", + "run_not_steerable", + ) + .into_response(); + } + RunStatus::Failed { .. } + | RunStatus::Succeeded { .. } + | RunStatus::Removing + | RunStatus::Dead => { + return terminal_control_response(&control); + } + RunStatus::Running => {} + } + // Steerability predicate. Best-effort, target-oriented: + // - If at least one API-mode session is active → forward. + // - Else if no agent stages are active at all → forward (worker hub buffers + // for the next session). + // - Else (active agents exist but all are non-steerable) → 409. + if managed_run.active_api_stages.is_empty() + && !managed_run.active_non_steerable_agent_stages.is_empty() + { + return ApiError::with_code( + StatusCode::CONFLICT, + "All currently running agent stages use a non-steerable backend.", + "agent_not_steerable", + ) + .into_response(); + } + if managed_run.active_api_stages.is_empty() && control.requires_active_api_session() + { + return ApiError::with_code( + StatusCode::CONFLICT, + "Run has no active API-mode agent session.", + "no_active_api_session", + ) + .into_response(); + } + Some(managed_run.answer_transport.clone()) } - RunStatus::Submitted - | RunStatus::Queued - | RunStatus::Starting - | RunStatus::Paused { .. } => { - return ApiError::with_code( - StatusCode::CONFLICT, - "Run is not currently running.", - "run_not_steerable", - ) - .into_response(); - } - RunStatus::Failed { .. } - | RunStatus::Succeeded { .. } - | RunStatus::Removing - | RunStatus::Dead => { - let code = if matches!(&control, RunControlRequest::Interrupt) { - "run_not_interruptible" - } else { - "run_not_steerable" - }; - return ApiError::with_code( - StatusCode::CONFLICT, - "Run is no longer steerable.", - code, - ) - .into_response(); - } - RunStatus::Running => {} + None => None, } - // Steerability predicate. Best-effort, target-oriented: - // - If at least one API-mode session is active → forward. - // - Else if no agent stages are active at all → forward (worker hub buffers - // for the next session). - // - Else (active agents exist but all are CLI-mode) → 409. - if managed_run.active_api_stages.is_empty() && !managed_run.active_cli_stages.is_empty() { - return ApiError::with_code( - StatusCode::CONFLICT, - "All currently running agent stages are CLI-mode and cannot be steered.", - "cli_agent_not_steerable", - ) - .into_response(); - } - if managed_run.active_api_stages.is_empty() && control.requires_active_api_session() { - return ApiError::with_code( - StatusCode::CONFLICT, - "Run has no active API-mode agent session.", - "no_active_api_session", - ) - .into_response(); - } - managed_run.answer_transport.clone() }; + let Some(answer_transport) = managed_answer_transport else { + return unmanaged_control_response(state.as_ref(), id, &control).await; + }; let Some(answer_transport) = answer_transport else { return ApiError::with_code( StatusCode::SERVICE_UNAVAILABLE, @@ -178,3 +179,32 @@ async fn control_run( .into_response(), } } + +fn terminal_control_response(control: &RunControlRequest) -> Response { + let code = if matches!(control, RunControlRequest::Interrupt) { + "run_not_interruptible" + } else { + "run_not_steerable" + }; + ApiError::with_code(StatusCode::CONFLICT, "Run is no longer steerable.", code).into_response() +} + +async fn unmanaged_control_response( + state: &AppState, + id: fabro_types::RunId, + control: &RunControlRequest, +) -> Response { + match durable_run_status(state, id).await { + Ok(Some(status)) if status.is_terminal() => terminal_control_response(control), + Ok(Some(_)) => ApiError::with_code( + StatusCode::SERVICE_UNAVAILABLE, + "Run has no live worker control channel.", + "worker_control_unavailable", + ) + .into_response(), + Ok(None) => ApiError::not_found("Run not found.").into_response(), + Err(err) => { + ApiError::new(StatusCode::INTERNAL_SERVER_ERROR, err.to_string()).into_response() + } + } +} diff --git a/lib/crates/fabro-server/src/server/tests.rs b/lib/crates/fabro-server/src/server/tests.rs index 0e0d7aa0a..d4dc471bd 100644 --- a/lib/crates/fabro-server/src/server/tests.rs +++ b/lib/crates/fabro-server/src/server/tests.rs @@ -6877,6 +6877,40 @@ async fn cancel_nonexistent_run_returns_not_found() { assert_status!(response, StatusCode::NOT_FOUND).await; } +#[tokio::test] +async fn cancel_terminal_durable_run_returns_conflict() { + let state = test_app_state(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = fixtures::RUN_1; + create_durable_run_with_events(&state, run_id, &[ + workflow_event::Event::WorkflowRunCompleted { + duration_ms: 1000, + artifact_count: 0, + status: "succeeded".to_string(), + reason: SuccessReason::Completed, + total_usd_micros: None, + final_git_commit_sha: None, + final_patch: None, + diff_summary: None, + billing: None, + }, + ]) + .await; + + let req = Request::builder() + .method("POST") + .uri(api(&format!("/runs/{run_id}/cancel"))) + .body(Body::empty()) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + let body = response_json!(response, StatusCode::CONFLICT).await; + assert_eq!( + body["errors"][0]["detail"], + "Run is already terminal and cannot be cancelled." + ); +} + #[tokio::test] async fn steer_nonexistent_run_returns_not_found() { let app = test_app_with(); @@ -6893,6 +6927,39 @@ async fn steer_nonexistent_run_returns_not_found() { assert_status!(response, StatusCode::NOT_FOUND).await; } +#[tokio::test] +async fn steer_terminal_durable_run_returns_run_not_steerable() { + let state = test_app_state(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = fixtures::RUN_1; + create_durable_run_with_events(&state, run_id, &[ + workflow_event::Event::WorkflowRunCompleted { + duration_ms: 1000, + artifact_count: 0, + status: "succeeded".to_string(), + reason: SuccessReason::Completed, + total_usd_micros: None, + final_git_commit_sha: None, + final_patch: None, + diff_summary: None, + billing: None, + }, + ]) + .await; + + let req = Request::builder() + .method("POST") + .uri(api(&format!("/runs/{run_id}/steer"))) + .header("content-type", "application/json") + .body(Body::from(r#"{"text":"try again"}"#)) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + let body = response_json!(response, StatusCode::CONFLICT).await; + assert_eq!(body["errors"][0]["code"], "run_not_steerable"); + assert_eq!(body["errors"][0]["detail"], "Run is no longer steerable."); +} + #[tokio::test] async fn steer_empty_text_returns_bad_request() { let state = test_app_state(); @@ -7164,6 +7231,150 @@ fn active_api_stage_projection_ignores_stale_deactivation() { ); } +fn acp_event_for_stage(run_id: &RunId, event: &workflow_event::Event) -> fabro_types::RunEvent { + workflow_event::to_run_event_at( + run_id, + event, + Utc::now(), + Some(&workflow_event::StageScope { + node_id: "agent".to_string(), + visit: 1, + parallel_group_id: None, + parallel_branch_id: None, + }), + ) +} + +#[tokio::test] +async fn steer_with_active_acp_stage_returns_non_steerable_conflict() { + let state = test_app_state(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = fixtures::RUN_1; + let (control_tx, _control_rx) = tokio::sync::mpsc::channel(1); + let _temp_dir = insert_running_control_run( + &state, + run_id, + Some(RunAnswerTransport::Subprocess { control_tx }), + ); + + let started = acp_event_for_stage(&run_id, &workflow_event::Event::AgentAcpStarted { + node_id: "agent".to_string(), + visit: 1, + mode: "acp".to_string(), + provider: "openai".to_string(), + model: "fake-acp".to_string(), + command: "python fake_agent.py".to_string(), + }); + update_live_run_from_event(&state, run_id, &started); + + let req = Request::builder() + .method("POST") + .uri(api(&format!("/runs/{run_id}/steer"))) + .header("content-type", "application/json") + .body(Body::from(r#"{"text":"try again"}"#)) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + assert_eq!(response.status(), StatusCode::CONFLICT); + let body = body_json(response.into_body()).await; + assert_eq!(body["errors"][0]["code"], "agent_not_steerable"); +} + +#[tokio::test] +async fn active_acp_stage_marker_clears_on_terminal_paths() { + let terminal_events: Vec = vec![ + workflow_event::Event::AgentAcpCompleted { + node_id: "agent".to_string(), + stdout: "done".to_string(), + stderr: String::new(), + stop_reason: "end_turn".to_string(), + duration_ms: 42, + }, + workflow_event::Event::AgentAcpCancelled { + node_id: "agent".to_string(), + stdout: "partial".to_string(), + stderr: "cancelled".to_string(), + duration_ms: 7, + }, + workflow_event::Event::AgentAcpTimedOut { + node_id: "agent".to_string(), + stdout: "partial".to_string(), + stderr: "timeout".to_string(), + duration_ms: 99, + }, + workflow_event::Event::StageCompleted { + node_id: "agent".to_string(), + name: "agent".to_string(), + index: 0, + duration_ms: 1, + status: "success".to_string(), + preferred_label: None, + suggested_next_ids: Vec::new(), + billing: None, + failure: None, + notes: None, + files_touched: Vec::new(), + context_updates: None, + jump_to_node: None, + context_values: None, + node_visits: None, + loop_failure_signatures: None, + restart_failure_signatures: None, + response: None, + attempt: 1, + max_attempts: 1, + }, + workflow_event::Event::StageFailed { + node_id: "agent".to_string(), + name: "agent".to_string(), + index: 0, + failure: FailureDetail::new("failed", FailureCategory::Deterministic), + will_retry: false, + duration_ms: 1, + billing: None, + actor: None, + }, + ]; + + for terminal_event in terminal_events { + let state = test_app_state(); + let app = crate::test_support::build_test_router(Arc::clone(&state)); + let run_id = fixtures::RUN_1; + let (control_tx, mut control_rx) = tokio::sync::mpsc::channel(1); + let _temp_dir = insert_running_control_run( + &state, + run_id, + Some(RunAnswerTransport::Subprocess { control_tx }), + ); + let started = acp_event_for_stage(&run_id, &workflow_event::Event::AgentAcpStarted { + node_id: "agent".to_string(), + visit: 1, + mode: "acp".to_string(), + provider: "openai".to_string(), + model: "fake-acp".to_string(), + command: "python fake_agent.py".to_string(), + }); + update_live_run_from_event(&state, run_id, &started); + let terminal = acp_event_for_stage(&run_id, &terminal_event); + update_live_run_from_event(&state, run_id, &terminal); + + let req = Request::builder() + .method("POST") + .uri(api(&format!("/runs/{run_id}/steer"))) + .header("content-type", "application/json") + .body(Body::from(r#"{"text":"try again"}"#)) + .unwrap(); + + let response = app.oneshot(req).await.unwrap(); + assert_status!(response, StatusCode::ACCEPTED).await; + let envelope = control_rx.recv().await.unwrap(); + assert!(matches!( + envelope.message, + WorkerControlMessage::Steer { ref text, .. } if text == "try again" + )); + } +} + #[tokio::test] async fn get_graph_returns_svg() { let state = test_app_state(); diff --git a/lib/crates/fabro-store/src/run_state.rs b/lib/crates/fabro-store/src/run_state.rs index d1bdf7b2a..176b32c5a 100644 --- a/lib/crates/fabro-store/src/run_state.rs +++ b/lib/crates/fabro-store/src/run_state.rs @@ -3,8 +3,9 @@ use std::str::FromStr; use chrono::{DateTime, Utc}; use fabro_types::run_event::{ - AgentCliStartedProps, AgentSessionActivatedProps, CheckpointCompletedProps, RunCompletedProps, - RunFailedProps, StageCompletedProps, StagePromptProps, + AgentAcpStartedProps, AgentCliStartedProps, AgentSessionActivatedProps, + CheckpointCompletedProps, RunCompletedProps, RunFailedProps, StageCompletedProps, + StagePromptProps, }; use fabro_types::settings::run::RunSandboxSettings; use fabro_types::{ @@ -371,6 +372,13 @@ impl RunProjectionReducer for RunProjection { }; stage.provider_used = Some(provider_used_from_agent_cli_started(props)); } + EventBody::AgentAcpStarted(props) => { + let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq) + else { + return Ok(()); + }; + stage.provider_used = Some(provider_used_from_agent_acp_started(props)); + } EventBody::CommandStarted(props) => { let script_invocation = serde_json::to_value(props).map_err(|err| { Error::InvalidEvent(format!("invalid command.started payload: {err}")) @@ -397,7 +405,8 @@ impl RunProjectionReducer for RunProjection { let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { return Ok(()); }; - apply_agent_cli_terminal( + apply_agent_terminal( + "agent.cli", stage, props, merge_agent_cli_output(&props.stdout, &props.stderr), @@ -408,7 +417,8 @@ impl RunProjectionReducer for RunProjection { let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { return Ok(()); }; - apply_agent_cli_terminal( + apply_agent_terminal( + "agent.cli", stage, props, merge_agent_cli_output(&props.stdout, &props.stderr), @@ -419,7 +429,44 @@ impl RunProjectionReducer for RunProjection { let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { return Ok(()); }; - apply_agent_cli_terminal( + apply_agent_terminal( + "agent.cli", + stage, + props, + merge_agent_cli_output(&props.stdout, &props.stderr), + CommandTermination::TimedOut, + )?; + } + EventBody::AgentAcpCompleted(props) => { + let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { + return Ok(()); + }; + apply_agent_terminal( + "agent.acp", + stage, + props, + merge_agent_cli_output(&props.stdout, &props.stderr), + CommandTermination::Exited, + )?; + } + EventBody::AgentAcpCancelled(props) => { + let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { + return Ok(()); + }; + apply_agent_terminal( + "agent.acp", + stage, + props, + merge_agent_cli_output(&props.stdout, &props.stderr), + CommandTermination::Cancelled, + )?; + } + EventBody::AgentAcpTimedOut(props) => { + let Some(stage) = stage_at_current_visit(self, stored, event.seq) else { + return Ok(()); + }; + apply_agent_terminal( + "agent.acp", stage, props, merge_agent_cli_output(&props.stdout, &props.stderr), @@ -837,25 +884,37 @@ fn provider_used_from_agent_session_activated(props: &AgentSessionActivatedProps } fn provider_used_from_agent_cli_started(props: &AgentCliStartedProps) -> Value { + provider_used_from_agent_process_started("cli", &props.provider, &props.model, &props.command) +} + +fn provider_used_from_agent_acp_started(props: &AgentAcpStartedProps) -> Value { + provider_used_from_agent_process_started("acp", &props.provider, &props.model, &props.command) +} + +fn provider_used_from_agent_process_started( + mode: &str, + provider: &str, + model: &str, + command: &str, +) -> Value { let mut provider_used = serde_json::Map::new(); - provider_used.insert("mode".to_string(), Value::String("cli".to_string())); - provider_used.insert( - "provider".to_string(), - Value::String(props.provider.clone()), - ); - provider_used.insert("model".to_string(), Value::String(props.model.clone())); - provider_used.insert("command".to_string(), Value::String(props.command.clone())); + provider_used.insert("mode".to_string(), Value::String(mode.to_string())); + provider_used.insert("provider".to_string(), Value::String(provider.to_string())); + provider_used.insert("model".to_string(), Value::String(model.to_string())); + provider_used.insert("command".to_string(), Value::String(command.to_string())); Value::Object(provider_used) } -fn apply_agent_cli_terminal( +fn apply_agent_terminal( + event_prefix: &str, stage: &mut StageProjection, props: &impl serde::Serialize, output: String, termination: CommandTermination, ) -> Result<()> { - let script_timing = serde_json::to_value(props) - .map_err(|err| Error::InvalidEvent(format!("invalid agent.cli terminal payload: {err}")))?; + let script_timing = serde_json::to_value(props).map_err(|err| { + Error::InvalidEvent(format!("invalid {event_prefix} terminal payload: {err}")) + })?; stage.output = Some(output); stage.termination = Some(termination); stage.script_timing = Some(script_timing); @@ -878,11 +937,13 @@ mod tests { use chrono::Utc; use fabro_types::run_event::run::RunFailedProps; use fabro_types::run_event::{ - AgentCliCancelledProps, AgentCliCompletedProps, AgentCliTimedOutProps, AgentMessageProps, - AgentSessionActivatedProps, AgentSessionEndedProps, AgentSessionStartedProps, - CheckpointCompletedProps, InterviewCompletedProps, InterviewOption, InterviewStartedProps, - RunControlEffectProps, StageCompletedProps, StageFailedProps, StagePromptProps, - StageRetryingProps, StageStartedProps, + AgentAcpCancelledProps, AgentAcpCompletedProps, AgentAcpStartedProps, + AgentAcpTimedOutProps, AgentCliCancelledProps, AgentCliCompletedProps, + AgentCliTimedOutProps, AgentMessageProps, AgentSessionActivatedProps, + AgentSessionEndedProps, AgentSessionStartedProps, CheckpointCompletedProps, + InterviewCompletedProps, InterviewOption, InterviewStartedProps, RunControlEffectProps, + StageCompletedProps, StageFailedProps, StagePromptProps, StageRetryingProps, + StageStartedProps, }; use fabro_types::{ BilledModelUsage, BilledTokenCounts, BlockedReason, Checkpoint, CheckpointRecord, @@ -1287,6 +1348,109 @@ mod tests { assert!(stage.provider_used.is_none()); } + #[test] + fn agent_acp_started_updates_stage_provider_used() { + let mut state = initialized_projection(); + let stage_id = StageId::new("code", 1); + start_stage(&mut state, &stage_id); + + state + .apply_event(&test_stage_event( + 4, + EventBody::AgentAcpStarted(AgentAcpStartedProps { + visit: 1, + mode: "acp".to_string(), + provider: "openai".to_string(), + model: "fake-acp".to_string(), + command: "python fake_agent.py".to_string(), + }), + stage_id.clone(), + )) + .unwrap(); + + let stage = state.stage(&stage_id).unwrap(); + assert_eq!( + stage.provider_used.as_ref().unwrap(), + &json!({ + "mode": "acp", + "provider": "openai", + "model": "fake-acp", + "command": "python fake_agent.py" + }) + ); + } + + #[test] + fn agent_acp_completed_updates_stage_output_projection() { + let mut state = initialized_projection(); + let stage_id = StageId::new("code", 1); + start_stage(&mut state, &stage_id); + + state + .apply_event(&test_stage_event( + 4, + EventBody::AgentAcpCompleted(AgentAcpCompletedProps { + stdout: "done".to_string(), + stderr: "warn".to_string(), + stop_reason: "end_turn".to_string(), + duration_ms: 42, + }), + stage_id.clone(), + )) + .unwrap(); + + let stage = state.stage(&stage_id).unwrap(); + assert_eq!(stage.output.as_deref(), Some("done\nwarn")); + assert_eq!(stage.termination, Some(CommandTermination::Exited)); + assert_eq!( + stage.script_timing.as_ref().unwrap()["stop_reason"], + serde_json::json!("end_turn") + ); + } + + #[test] + fn agent_acp_cancelled_and_timed_out_update_terminal_projection() { + let mut cancelled = initialized_projection(); + let cancelled_stage_id = StageId::new("cancelled", 1); + start_stage(&mut cancelled, &cancelled_stage_id); + + cancelled + .apply_event(&test_stage_event( + 4, + EventBody::AgentAcpCancelled(AgentAcpCancelledProps { + stdout: "partial".to_string(), + stderr: "cancelled".to_string(), + duration_ms: 7, + }), + cancelled_stage_id.clone(), + )) + .unwrap(); + + let stage = cancelled.stage(&cancelled_stage_id).unwrap(); + assert_eq!(stage.output.as_deref(), Some("partial\ncancelled")); + assert_eq!(stage.termination, Some(CommandTermination::Cancelled)); + + let mut timed_out = initialized_projection(); + let timed_out_stage_id = StageId::new("timed_out", 1); + start_stage(&mut timed_out, &timed_out_stage_id); + + timed_out + .apply_event(&test_stage_event( + 4, + EventBody::AgentAcpTimedOut(AgentAcpTimedOutProps { + stdout: "partial".to_string(), + stderr: "timeout".to_string(), + duration_ms: 99, + }), + timed_out_stage_id.clone(), + )) + .unwrap(); + + let stage = timed_out.stage(&timed_out_stage_id).unwrap(); + assert_eq!(stage.output.as_deref(), Some("partial\ntimeout")); + assert_eq!(stage.termination, Some(CommandTermination::TimedOut)); + } + #[test] fn agent_cli_completed_updates_stage_output_projection() { let mut state = initialized_projection(); diff --git a/lib/crates/fabro-test/src/lib.rs b/lib/crates/fabro-test/src/lib.rs index 77841bce7..8d90c1a3f 100644 --- a/lib/crates/fabro-test/src/lib.rs +++ b/lib/crates/fabro-test/src/lib.rs @@ -152,6 +152,43 @@ pub fn apply_test_isolation(cmd: &mut std::process::Command, home_dir: &Path) { apply_test_isolation_with_lookup(cmd, home_dir, |name| std::env::var_os(name)); } +#[must_use] +pub fn isolated_env(home_dir: &Path) -> HashMap { + let mut env = HashMap::new(); + if let Some(coverage) = + std::env::var_os(EnvVars::LLVM_PROFILE_FILE).and_then(|value| value.into_string().ok()) + { + env.insert(EnvVars::LLVM_PROFILE_FILE.to_string(), coverage); + } + if let Some(path) = std::env::var_os(EnvVars::PATH).and_then(|value| value.into_string().ok()) { + env.insert(EnvVars::PATH.to_string(), path); + } + env.insert(EnvVars::NO_COLOR.to_string(), "1".to_string()); + env.insert(EnvVars::HOME.to_string(), home_dir.display().to_string()); + env.insert( + EnvVars::FABRO_NO_UPGRADE_CHECK.to_string(), + "true".to_string(), + ); + env.insert( + EnvVars::FABRO_HTTP_PROXY_POLICY.to_string(), + "disabled".to_string(), + ); + env.insert(EnvVars::FABRO_TELEMETRY.to_string(), "off".to_string()); + env.insert( + EnvVars::FABRO_SUPPRESS_OPEN_BROWSER.to_string(), + "1".to_string(), + ); + env.insert( + EnvVars::FABRO_SERVER_MAX_CONCURRENT_RUNS.to_string(), + "64".to_string(), + ); + env.insert( + EnvVars::FABRO_TEST_IN_MEMORY_STORE.to_string(), + "1".to_string(), + ); + env +} + fn apply_test_isolation_with_lookup( cmd: &mut std::process::Command, home_dir: &Path, diff --git a/lib/crates/fabro-types/src/graph.rs b/lib/crates/fabro-types/src/graph.rs index 90e9307f4..786fe269e 100644 --- a/lib/crates/fabro-types/src/graph.rs +++ b/lib/crates/fabro-types/src/graph.rs @@ -3,6 +3,8 @@ use std::time::Duration; use serde::{Deserialize, Serialize}; +use crate::LlmBackend; + /// Typed attribute values for nodes, edges, and graph-level attributes. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub enum AttrValue { @@ -266,6 +268,16 @@ impl Node { self.str_attr("backend") } + #[must_use] + pub fn llm_backend(&self) -> Option> { + self.backend().map(str::parse) + } + + #[must_use] + pub fn acp_command(&self) -> Option<&str> { + self.str_attr("acp_command") + } + #[must_use] pub fn selection(&self) -> &str { self.str_attr("selection").unwrap_or("deterministic") diff --git a/lib/crates/fabro-types/src/lib.rs b/lib/crates/fabro-types/src/lib.rs index dcac895f2..e4da87ec6 100644 --- a/lib/crates/fabro-types/src/lib.rs +++ b/lib/crates/fabro-types/src/lib.rs @@ -13,6 +13,7 @@ pub mod event_envelope; pub mod failure_signature; pub mod graph; pub mod interview; +pub mod llm_backend; pub mod outcome; pub mod principal; pub mod pull_request; @@ -57,6 +58,7 @@ pub use graph::{ shape_to_handler_type, }; pub use interview::{InterviewQuestionRecord, QuestionType}; +pub use llm_backend::LlmBackend; pub use outcome::{ FailureCategory, FailureDetail, NodeResult, Outcome, OutcomeMeta, StageOutcome, StageState, }; @@ -102,5 +104,6 @@ pub use stage_id::{InvalidStageVisit, ParallelBranchId, StageId}; pub use start::StartRecord; pub use status::{ BlockedReason, FailureReason, InvalidTransition, ParseFailureReasonError, - ParseSuccessReasonError, RunControlAction, RunStatus, SuccessReason, TerminalStatus, + ParseSuccessReasonError, RunControlAction, RunStatus, RunStatusKind, SuccessReason, + TerminalStatus, }; diff --git a/lib/crates/fabro-types/src/llm_backend.rs b/lib/crates/fabro-types/src/llm_backend.rs new file mode 100644 index 000000000..5a91fce2e --- /dev/null +++ b/lib/crates/fabro-types/src/llm_backend.rs @@ -0,0 +1,32 @@ +use serde::{Deserialize, Serialize}; +use strum::{Display, EnumString, IntoStaticStr, VariantArray, VariantNames}; + +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + Hash, + Serialize, + Deserialize, + Display, + EnumString, + IntoStaticStr, + VariantArray, + VariantNames, +)] +#[serde(rename_all = "snake_case")] +#[strum(serialize_all = "snake_case")] +pub enum LlmBackend { + Api, + Cli, + Acp, +} + +impl LlmBackend { + #[must_use] + pub fn expected_values() -> String { + ::VARIANTS.join(", ") + } +} diff --git a/lib/crates/fabro-types/src/run_event/misc.rs b/lib/crates/fabro-types/src/run_event/misc.rs index 86dba4f64..22c1f340e 100644 --- a/lib/crates/fabro-types/src/run_event/misc.rs +++ b/lib/crates/fabro-types/src/run_event/misc.rs @@ -335,6 +335,37 @@ pub struct AgentCliTimedOutProps { pub duration_ms: u64, } +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct AgentAcpStartedProps { + pub visit: u32, + pub mode: String, + pub provider: String, + pub model: String, + pub command: String, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct AgentAcpCompletedProps { + pub stdout: String, + pub stderr: String, + pub stop_reason: String, + pub duration_ms: u64, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct AgentAcpCancelledProps { + pub stdout: String, + pub stderr: String, + pub duration_ms: u64, +} + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct AgentAcpTimedOutProps { + pub stdout: String, + pub stderr: String, + pub duration_ms: u64, +} + #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct PullRequestCreatedProps { pub pr_url: String, diff --git a/lib/crates/fabro-types/src/run_event/mod.rs b/lib/crates/fabro-types/src/run_event/mod.rs index d9887e016..b473ad6b4 100644 --- a/lib/crates/fabro-types/src/run_event/mod.rs +++ b/lib/crates/fabro-types/src/run_event/mod.rs @@ -290,6 +290,14 @@ pub enum EventBody { AgentCliCancelled(AgentCliCancelledProps), #[serde(rename = "agent.cli.timed_out")] AgentCliTimedOut(AgentCliTimedOutProps), + #[serde(rename = "agent.acp.started")] + AgentAcpStarted(AgentAcpStartedProps), + #[serde(rename = "agent.acp.completed")] + AgentAcpCompleted(AgentAcpCompletedProps), + #[serde(rename = "agent.acp.cancelled")] + AgentAcpCancelled(AgentAcpCancelledProps), + #[serde(rename = "agent.acp.timed_out")] + AgentAcpTimedOut(AgentAcpTimedOutProps), #[serde(rename = "pull_request.created")] PullRequestCreated(PullRequestCreatedProps), #[serde(rename = "pull_request.failed")] @@ -484,6 +492,10 @@ impl EventBody { Self::AgentCliCompleted(_) => "agent.cli.completed", Self::AgentCliCancelled(_) => "agent.cli.cancelled", Self::AgentCliTimedOut(_) => "agent.cli.timed_out", + Self::AgentAcpStarted(_) => "agent.acp.started", + Self::AgentAcpCompleted(_) => "agent.acp.completed", + Self::AgentAcpCancelled(_) => "agent.acp.cancelled", + Self::AgentAcpTimedOut(_) => "agent.acp.timed_out", Self::PullRequestCreated(_) => "pull_request.created", Self::PullRequestFailed(_) => "pull_request.failed", Self::DevcontainerResolved(_) => "devcontainer.resolved", @@ -629,6 +641,12 @@ fn is_known_event_name(event: &str) -> bool { | "command.completed" | "agent.cli.started" | "agent.cli.completed" + | "agent.cli.cancelled" + | "agent.cli.timed_out" + | "agent.acp.started" + | "agent.acp.completed" + | "agent.acp.cancelled" + | "agent.acp.timed_out" | "pull_request.created" | "pull_request.failed" | "devcontainer.resolved" diff --git a/lib/crates/fabro-types/src/settings/run.rs b/lib/crates/fabro-types/src/settings/run.rs index 478feda2a..0650e6b39 100644 --- a/lib/crates/fabro-types/src/settings/run.rs +++ b/lib/crates/fabro-types/src/settings/run.rs @@ -7,6 +7,7 @@ //! behavior, and artifact collection. use std::collections::HashMap; +use std::path::PathBuf; use std::time::Duration as StdDuration; use serde::ser::SerializeStruct; @@ -341,6 +342,10 @@ pub struct RunAgentSettings { pub struct McpServerSettings { pub name: String, pub transport: McpTransport, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub current_dir: Option, + #[serde(default, skip_serializing_if = "is_false")] + pub clear_env: bool, pub startup_timeout_secs: u64, pub tool_timeout_secs: u64, } @@ -353,6 +358,8 @@ impl Default for McpServerSettings { command: Vec::new(), env: HashMap::new(), }, + current_dir: None, + clear_env: false, startup_timeout_secs: 10, tool_timeout_secs: 60, } @@ -389,6 +396,14 @@ pub enum McpTransport { }, } +#[expect( + clippy::trivially_copy_pass_by_ref, + reason = "serde skip_serializing_if helpers receive borrowed field values" +)] +fn is_false(value: &bool) -> bool { + !*value +} + #[derive(Debug, Clone, Copy, Deserialize, PartialEq, Eq, Default, Serialize)] #[serde(rename_all = "snake_case")] pub enum TlsMode { diff --git a/lib/crates/fabro-types/src/status.rs b/lib/crates/fabro-types/src/status.rs index d59f89b65..cf2d7d1bb 100644 --- a/lib/crates/fabro-types/src/status.rs +++ b/lib/crates/fabro-types/src/status.rs @@ -2,6 +2,35 @@ use std::fmt; use std::str::FromStr; use serde::{Deserialize, Serialize}; +use strum::{Display, EnumString, IntoStaticStr}; + +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + Hash, + Serialize, + Deserialize, + Display, + EnumString, + IntoStaticStr, +)] +#[serde(rename_all = "snake_case")] +#[strum(serialize_all = "snake_case")] +pub enum RunStatusKind { + Submitted, + Queued, + Starting, + Running, + Blocked, + Paused, + Removing, + Succeeded, + Failed, + Dead, +} #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] #[serde(tag = "kind", rename_all = "snake_case")] @@ -19,6 +48,10 @@ pub enum RunStatus { } impl RunStatus { + pub fn kind(self) -> RunStatusKind { + self.into() + } + /// Whether the run has reached a terminal outcome and stops poll loops, /// finalization, and similar "done" handling. pub fn is_terminal(self) -> bool { @@ -138,6 +171,23 @@ impl RunStatus { } } +impl From for RunStatusKind { + fn from(status: RunStatus) -> Self { + match status { + RunStatus::Submitted => Self::Submitted, + RunStatus::Queued => Self::Queued, + RunStatus::Starting => Self::Starting, + RunStatus::Running => Self::Running, + RunStatus::Blocked { .. } => Self::Blocked, + RunStatus::Paused { .. } => Self::Paused, + RunStatus::Removing => Self::Removing, + RunStatus::Succeeded { .. } => Self::Succeeded, + RunStatus::Failed { .. } => Self::Failed, + RunStatus::Dead => Self::Dead, + } + } +} + impl fmt::Display for RunStatus { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { diff --git a/lib/crates/fabro-validate/Cargo.toml b/lib/crates/fabro-validate/Cargo.toml index 0dc049b04..2ea44ae2d 100644 --- a/lib/crates/fabro-validate/Cargo.toml +++ b/lib/crates/fabro-validate/Cargo.toml @@ -15,5 +15,6 @@ workspace = true [dependencies] fabro-graphviz = { path = "../fabro-graphviz" } fabro-model = { path = "../fabro-model" } +fabro-types = { path = "../fabro-types" } serde = { workspace = true } -thiserror = { workspace = true } \ No newline at end of file +thiserror = { workspace = true } diff --git a/lib/crates/fabro-validate/src/rules/backend_valid.rs b/lib/crates/fabro-validate/src/rules/backend_valid.rs new file mode 100644 index 000000000..9a04af737 --- /dev/null +++ b/lib/crates/fabro-validate/src/rules/backend_valid.rs @@ -0,0 +1,140 @@ +use fabro_graphviz::graph::{Graph, Node}; +use fabro_types::LlmBackend; + +use crate::{Diagnostic, LintRule, Severity}; + +pub(super) fn rule() -> Box { + Box::new(Rule) +} + +struct Rule; + +impl LintRule for Rule { + fn name(&self) -> &'static str { + "backend_valid" + } + + fn apply(&self, graph: &Graph) -> Vec { + let mut diagnostics = Vec::new(); + for node in graph.nodes.values() { + if let Some(backend) = node.backend() { + match node.llm_backend() { + Some(Err(_)) => { + let expected = LlmBackend::expected_values(); + diagnostics.push(Diagnostic { + rule: self.name().to_string(), + severity: Severity::Error, + message: format!( + "unsupported LLM backend \"{backend}\"; expected one of: {expected}" + ), + node_id: Some(node.id.clone()), + edge: None, + fix: Some(format!("Use one of: {expected}")), + }); + } + Some(Ok(LlmBackend::Acp)) if acp_command_missing(node) => { + diagnostics.push(Diagnostic { + rule: self.name().to_string(), + severity: Severity::Error, + message: "backend=\"acp\" requires acp_command because Fabro does \ + not install ACP agents" + .to_string(), + node_id: Some(node.id.clone()), + edge: None, + fix: Some( + "Set acp_command to a stdio ACP command available in the sandbox" + .to_string(), + ), + }); + } + Some(Ok(_)) | None => {} + } + } + } + diagnostics + } +} + +fn acp_command_missing(node: &Node) -> bool { + match node.acp_command() { + Some(command) => command.trim().is_empty(), + None => true, + } +} + +#[cfg(test)] +mod tests { + use fabro_graphviz::graph::{AttrValue, Node}; + + use super::Rule; + use crate::rules::test_support::minimal_graph; + use crate::{LintRule, Severity}; + + #[test] + fn backend_valid_accepts_absent_api_and_cli() { + for backend in [None, Some("api"), Some("cli")] { + let mut graph = minimal_graph(); + let mut node = Node::new("work"); + if let Some(backend) = backend { + node.attrs.insert( + "backend".to_string(), + AttrValue::String(backend.to_string()), + ); + } + graph.nodes.insert("work".to_string(), node); + + assert!(Rule.apply(&graph).is_empty(), "backend: {backend:?}"); + } + } + + #[test] + fn backend_valid_rejects_unknown_backend() { + let mut graph = minimal_graph(); + let mut node = Node::new("work"); + node.attrs.insert( + "backend".to_string(), + AttrValue::String("codex".to_string()), + ); + graph.nodes.insert("work".to_string(), node); + + let diagnostics = Rule.apply(&graph); + assert_eq!(diagnostics.len(), 1); + assert_eq!(diagnostics[0].severity, Severity::Error); + assert!( + diagnostics[0] + .message + .contains("unsupported LLM backend \"codex\"; expected one of: api, cli, acp") + ); + } + + #[test] + fn backend_valid_requires_acp_command_for_acp_backend() { + let mut graph = minimal_graph(); + let mut node = Node::new("work"); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + graph.nodes.insert("work".to_string(), node); + + let diagnostics = Rule.apply(&graph); + assert_eq!(diagnostics.len(), 1); + assert_eq!(diagnostics[0].severity, Severity::Error); + assert!(diagnostics[0].message.contains( + "backend=\"acp\" requires acp_command because Fabro does not install ACP agents" + )); + } + + #[test] + fn backend_valid_accepts_acp_backend_with_acp_command() { + let mut graph = minimal_graph(); + let mut node = Node::new("work"); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String("agent-acp".to_string()), + ); + graph.nodes.insert("work".to_string(), node); + + assert!(Rule.apply(&graph).is_empty()); + } +} diff --git a/lib/crates/fabro-validate/src/rules/mod.rs b/lib/crates/fabro-validate/src/rules/mod.rs index 1d01da29f..208fe0bcf 100644 --- a/lib/crates/fabro-validate/src/rules/mod.rs +++ b/lib/crates/fabro-validate/src/rules/mod.rs @@ -1,4 +1,5 @@ mod all_conditional_edges; +mod backend_valid; mod condition_syntax; mod direction_valid; mod edge_target_exists; @@ -43,6 +44,7 @@ pub fn built_in_rules() -> Vec> { condition_syntax::rule(), stylesheet_syntax::rule(), type_known::rule(), + backend_valid::rule(), fidelity_valid::rule(), retry_target_exists::rule(), goal_gate_has_retry::rule(), diff --git a/lib/crates/fabro-workflow/Cargo.toml b/lib/crates/fabro-workflow/Cargo.toml index 91dd3dac0..d535328b0 100644 --- a/lib/crates/fabro-workflow/Cargo.toml +++ b/lib/crates/fabro-workflow/Cargo.toml @@ -19,6 +19,7 @@ workspace = true [dependencies] anyhow.workspace = true fabro-auth = { path = "../fabro-auth" } +fabro-acp = { path = "../fabro-acp" } fabro-agent = { path = "../fabro-agent" } fabro-config = { path = "../fabro-config" } fabro-graphviz = { path = "../fabro-graphviz" } @@ -70,7 +71,8 @@ toml.workspace = true fabro-vault = { path = "../fabro-vault" } [dev-dependencies] base64.workspace = true -fabro-sandbox = { path = "../fabro-sandbox", features = ["daytona", "test-support"] } +fabro-acp = { path = "../fabro-acp", features = ["test-support"] } +fabro-sandbox = { path = "../fabro-sandbox", features = ["daytona", "docker", "test-support"] } fabro-mcp = { path = "../fabro-mcp" } tokio = { workspace = true, features = ["test-util", "macros"] } object_store.workspace = true diff --git a/lib/crates/fabro-workflow/src/event/convert.rs b/lib/crates/fabro-workflow/src/event/convert.rs index 6495b24e6..c05135c1a 100644 --- a/lib/crates/fabro-workflow/src/event/convert.rs +++ b/lib/crates/fabro-workflow/src/event/convert.rs @@ -986,38 +986,6 @@ fn event_body_from_event(event: &Event) -> EventBody { to_model: to_model.clone(), error: error.clone(), }), - Event::CliEnsureStarted { cli_name, provider } => { - EventBody::CliEnsureStarted(fabro_types::CliEnsureStartedProps { - cli_name: cli_name.clone(), - provider: provider.clone(), - }) - } - Event::CliEnsureCompleted { - cli_name, - provider, - already_installed, - node_installed, - duration_ms, - } => EventBody::CliEnsureCompleted(fabro_types::CliEnsureCompletedProps { - cli_name: cli_name.clone(), - provider: provider.clone(), - already_installed: *already_installed, - node_installed: *node_installed, - duration_ms: *duration_ms, - }), - Event::CliEnsureFailed { - cli_name, - provider, - error, - duration_ms, - exec_output_tail, - } => EventBody::CliEnsureFailed(fabro_types::CliEnsureFailedProps { - cli_name: cli_name.clone(), - provider: provider.clone(), - error: error.clone(), - duration_ms: *duration_ms, - exec_output_tail: exec_output_tail.clone(), - }), Event::CommandStarted { script, command, @@ -1134,6 +1102,52 @@ fn event_body_from_event(event: &Event) -> EventBody { stderr: stderr.clone(), duration_ms: *duration_ms, }), + Event::AgentAcpStarted { + visit, + mode, + provider, + model, + command, + .. + } => EventBody::AgentAcpStarted(fabro_types::AgentAcpStartedProps { + visit: *visit, + mode: mode.clone(), + provider: provider.clone(), + model: model.clone(), + command: command.clone(), + }), + Event::AgentAcpCompleted { + stdout, + stderr, + stop_reason, + duration_ms, + .. + } => EventBody::AgentAcpCompleted(fabro_types::AgentAcpCompletedProps { + stdout: stdout.clone(), + stderr: stderr.clone(), + stop_reason: stop_reason.clone(), + duration_ms: *duration_ms, + }), + Event::AgentAcpCancelled { + stdout, + stderr, + duration_ms, + .. + } => EventBody::AgentAcpCancelled(fabro_types::AgentAcpCancelledProps { + stdout: stdout.clone(), + stderr: stderr.clone(), + duration_ms: *duration_ms, + }), + Event::AgentAcpTimedOut { + stdout, + stderr, + duration_ms, + .. + } => EventBody::AgentAcpTimedOut(fabro_types::AgentAcpTimedOutProps { + stdout: stdout.clone(), + stderr: stderr.clone(), + duration_ms: *duration_ms, + }), Event::PullRequestCreated { pr_url, pr_number, @@ -2009,6 +2023,110 @@ mod tests { } } + #[test] + fn agent_acp_events_map_to_event_bodies_with_stage_scope() { + let scope = StageScope { + node_id: "code".to_string(), + visit: 2, + parallel_group_id: Some(StageId::new("fanout", 1)), + parallel_branch_id: Some(ParallelBranchId::new(StageId::new("fanout", 1), 0)), + }; + + let started = to_run_event_at( + &fixtures::RUN_1, + &Event::AgentAcpStarted { + node_id: "code".to_string(), + visit: 2, + mode: "acp".to_string(), + provider: "openai".to_string(), + model: "fake-acp".to_string(), + command: "python fake_agent.py".to_string(), + }, + Utc::now(), + Some(&scope), + ); + assert_eq!(started.event_name(), "agent.acp.started"); + assert_eq!(started.node_id.as_deref(), Some("code")); + assert_eq!(started.stage_id, Some(StageId::new("code", 2))); + assert_eq!(started.parallel_group_id, scope.parallel_group_id); + assert_eq!(started.parallel_branch_id, scope.parallel_branch_id); + match &started.body { + EventBody::AgentAcpStarted(props) => { + assert_eq!(props.visit, 2); + assert_eq!(props.mode, "acp"); + assert_eq!(props.provider, "openai"); + assert_eq!(props.model, "fake-acp"); + assert_eq!(props.command, "python fake_agent.py"); + } + other => panic!("expected AgentAcpStarted, got {other:?}"), + } + + let completed = to_run_event_at( + &fixtures::RUN_1, + &Event::AgentAcpCompleted { + node_id: "code".to_string(), + stdout: "done".to_string(), + stderr: "warn".to_string(), + stop_reason: "end_turn".to_string(), + duration_ms: 42, + }, + Utc::now(), + Some(&scope), + ); + assert_eq!(completed.event_name(), "agent.acp.completed"); + match &completed.body { + EventBody::AgentAcpCompleted(props) => { + assert_eq!(props.stdout, "done"); + assert_eq!(props.stderr, "warn"); + assert_eq!(props.stop_reason, "end_turn"); + assert_eq!(props.duration_ms, 42); + } + other => panic!("expected AgentAcpCompleted, got {other:?}"), + } + + let cancelled = to_run_event_at( + &fixtures::RUN_1, + &Event::AgentAcpCancelled { + node_id: "code".to_string(), + stdout: "partial".to_string(), + stderr: "cancelled".to_string(), + duration_ms: 7, + }, + Utc::now(), + Some(&scope), + ); + assert_eq!(cancelled.event_name(), "agent.acp.cancelled"); + assert_eq!(cancelled.stage_id, Some(StageId::new("code", 2))); + assert!(matches!( + cancelled.body, + EventBody::AgentAcpCancelled(fabro_types::AgentAcpCancelledProps { + duration_ms: 7, + .. + }) + )); + + let timed_out = to_run_event_at( + &fixtures::RUN_1, + &Event::AgentAcpTimedOut { + node_id: "code".to_string(), + stdout: "partial".to_string(), + stderr: "timeout".to_string(), + duration_ms: 99, + }, + Utc::now(), + Some(&scope), + ); + assert_eq!(timed_out.event_name(), "agent.acp.timed_out"); + assert_eq!(timed_out.stage_id, Some(StageId::new("code", 2))); + assert!(matches!( + timed_out.body, + EventBody::AgentAcpTimedOut(fabro_types::AgentAcpTimedOutProps { + duration_ms: 99, + .. + }) + )); + } + #[test] fn stall_watchdog_timeout_populates_watchdog_actor() { let stored = to_run_event(&fixtures::RUN_1, &Event::StallWatchdogTimeout { diff --git a/lib/crates/fabro-workflow/src/event/events.rs b/lib/crates/fabro-workflow/src/event/events.rs index 9d3a76cad..f2067c5f4 100644 --- a/lib/crates/fabro-workflow/src/event/events.rs +++ b/lib/crates/fabro-workflow/src/event/events.rs @@ -495,25 +495,6 @@ pub enum Event { to_model: String, error: String, }, - CliEnsureStarted { - cli_name: String, - provider: String, - }, - CliEnsureCompleted { - cli_name: String, - provider: String, - already_installed: bool, - node_installed: bool, - duration_ms: u64, - }, - CliEnsureFailed { - cli_name: String, - provider: String, - error: String, - duration_ms: u64, - #[serde(default, skip_serializing_if = "Option::is_none")] - exec_output_tail: Option, - }, CommandStarted { node_id: String, script: String, @@ -621,6 +602,33 @@ pub enum Event { stderr: String, duration_ms: u64, }, + AgentAcpStarted { + node_id: String, + visit: u32, + mode: String, + provider: String, + model: String, + command: String, + }, + AgentAcpCompleted { + node_id: String, + stdout: String, + stderr: String, + stop_reason: String, + duration_ms: u64, + }, + AgentAcpCancelled { + node_id: String, + stdout: String, + stderr: String, + duration_ms: u64, + }, + AgentAcpTimedOut { + node_id: String, + stdout: String, + stderr: String, + duration_ms: u64, + }, PullRequestCreated { pr_url: String, pr_number: u64, @@ -1241,48 +1249,6 @@ impl Event { "LLM provider failover" ); } - Self::CliEnsureStarted { - cli_name, provider, .. - } => { - debug!(cli_name, provider, "CLI ensure started"); - } - Self::CliEnsureCompleted { - cli_name, - provider, - already_installed, - node_installed, - duration_ms, - } => { - info!( - cli_name, - provider, - already_installed, - node_installed, - duration_ms, - "CLI ensure completed" - ); - } - Self::CliEnsureFailed { - cli_name, - provider, - error, - duration_ms, - exec_output_tail, - } => { - let tail = fabro_types::ExecOutputTail::trace_summary(exec_output_tail.as_ref()); - error!( - cli_name, - provider, - error, - duration_ms, - exec_output_tail_present = tail.present, - exec_stdout_tail_bytes = tail.stdout_bytes, - exec_stderr_tail_bytes = tail.stderr_bytes, - exec_stdout_truncated = tail.stdout_truncated, - exec_stderr_truncated = tail.stderr_truncated, - "CLI ensure failed" - ); - } Self::CommandStarted { node_id, language, @@ -1378,6 +1344,36 @@ impl Event { } => { debug!(node_id, duration_ms, "Agent CLI timed out"); } + Self::AgentAcpStarted { + node_id, + provider, + model, + .. + } => { + debug!(node_id, provider, model, "Agent ACP started"); + } + Self::AgentAcpCompleted { + node_id, + stop_reason, + duration_ms, + .. + } => { + debug!(node_id, stop_reason, duration_ms, "Agent ACP completed"); + } + Self::AgentAcpCancelled { + node_id, + duration_ms, + .. + } => { + debug!(node_id, duration_ms, "Agent ACP cancelled"); + } + Self::AgentAcpTimedOut { + node_id, + duration_ms, + .. + } => { + debug!(node_id, duration_ms, "Agent ACP timed out"); + } Self::PullRequestCreated { pr_url, pr_number, diff --git a/lib/crates/fabro-workflow/src/event/names.rs b/lib/crates/fabro-workflow/src/event/names.rs index a4a630b8f..1607975f5 100644 --- a/lib/crates/fabro-workflow/src/event/names.rs +++ b/lib/crates/fabro-workflow/src/event/names.rs @@ -121,9 +121,6 @@ pub fn event_name(event: &Event) -> &'static str { Event::ArtifactCaptured { .. } => "artifact.captured", Event::SshAccessReady { .. } => "ssh.ready", Event::Failover { .. } => "agent.failover", - Event::CliEnsureStarted { .. } => "cli.ensure.started", - Event::CliEnsureCompleted { .. } => "cli.ensure.completed", - Event::CliEnsureFailed { .. } => "cli.ensure.failed", Event::CommandStarted { .. } => "command.started", Event::CommandCompleted { .. } => "command.completed", Event::AgentCliStarted { .. } => "agent.cli.started", @@ -137,6 +134,10 @@ pub fn event_name(event: &Event) -> &'static str { Event::AgentSteerDropped { .. } => "agent.steer.dropped", Event::AgentCliCancelled { .. } => "agent.cli.cancelled", Event::AgentCliTimedOut { .. } => "agent.cli.timed_out", + Event::AgentAcpStarted { .. } => "agent.acp.started", + Event::AgentAcpCompleted { .. } => "agent.acp.completed", + Event::AgentAcpCancelled { .. } => "agent.acp.cancelled", + Event::AgentAcpTimedOut { .. } => "agent.acp.timed_out", Event::PullRequestCreated { .. } => "pull_request.created", Event::PullRequestFailed { .. } => "pull_request.failed", Event::DevcontainerResolved { .. } => "devcontainer.resolved", diff --git a/lib/crates/fabro-workflow/src/event/stored_fields.rs b/lib/crates/fabro-workflow/src/event/stored_fields.rs index afa202c1f..20abd7a96 100644 --- a/lib/crates/fabro-workflow/src/event/stored_fields.rs +++ b/lib/crates/fabro-workflow/src/event/stored_fields.rs @@ -122,7 +122,20 @@ fn stored_event_fields_for_variant(event: &Event) -> StoredEventFields { | Event::AgentCliStarted { node_id, .. } | Event::AgentCliCompleted { node_id, .. } | Event::AgentCliCancelled { node_id, .. } - | Event::AgentCliTimedOut { node_id, .. } => node_stored_fields(Some(node_id.clone())), + | Event::AgentCliTimedOut { node_id, .. } + | Event::AgentAcpCompleted { node_id, .. } + | Event::AgentAcpCancelled { node_id, .. } + | Event::AgentAcpTimedOut { node_id, .. } => node_stored_fields(Some(node_id.clone())), + Event::AgentAcpStarted { node_id, visit, .. } => { + let node_id_str = node_id.clone(); + let node_label = default_node_label(Some(&node_id_str), None); + StoredEventFields { + node_id: Some(node_id_str.clone()), + node_label, + stage_id: Some(StageId::new(node_id_str, *visit)), + ..StoredEventFields::default() + } + } Event::AgentSessionStarted { session_id, parent_session_id, diff --git a/lib/crates/fabro-workflow/src/handler/agent.rs b/lib/crates/fabro-workflow/src/handler/agent.rs index 1be5cb25e..376d45431 100644 --- a/lib/crates/fabro-workflow/src/handler/agent.rs +++ b/lib/crates/fabro-workflow/src/handler/agent.rs @@ -28,35 +28,35 @@ pub enum CodergenResult { Full(Outcome), } +pub struct CodergenRunRequest<'a> { + pub node: &'a Node, + pub prompt: &'a str, + pub context: &'a Context, + pub thread_id: Option<&'a str>, + pub emitter: &'a Arc, + pub sandbox: &'a Arc, + pub tool_hooks: Option>, + pub cancel_token: CancellationToken, +} + +pub struct OneShotRequest<'a> { + pub node: &'a Node, + pub prompt: &'a str, + pub system_prompt: Option<&'a str>, + pub emitter: &'a Arc, + pub stage_scope: &'a StageScope, + pub sandbox: &'a Arc, + pub cancel_token: CancellationToken, +} + /// Backend interface for LLM execution in codergen nodes. -#[allow( - clippy::too_many_arguments, - reason = "Codergen backends need the node, prompt, context, and runtime handles separately." -)] #[async_trait] pub trait CodergenBackend: Send + Sync { /// Run a multi-turn agent loop (the default codergen mode). - async fn run( - &self, - node: &Node, - prompt: &str, - context: &Context, - thread_id: Option<&str>, - emitter: &Arc, - sandbox: &Arc, - tool_hooks: Option>, - cancel_token: CancellationToken, - ) -> Result; + async fn run(&self, request: CodergenRunRequest<'_>) -> Result; /// Run a single LLM call with no tools (one_shot mode). - async fn one_shot( - &self, - _node: &Node, - _prompt: &str, - _system_prompt: Option<&str>, - _emitter: &Arc, - _stage_scope: &StageScope, - ) -> Result { + async fn one_shot(&self, _request: OneShotRequest<'_>) -> Result { Err(Error::Validation( "one_shot mode not supported by this backend".into(), )) @@ -301,16 +301,16 @@ impl Handler for AgentHandler { let (response_text, stage_usage, backend_files_touched, last_file_touched) = if let Some(backend) = &self.backend { let result = backend - .run( + .run(CodergenRunRequest { node, - &prompt, + prompt: &prompt, context, - thread_id.as_deref(), - &services.run.emitter, - &services.run.sandbox, + thread_id: thread_id.as_deref(), + emitter: &services.run.emitter, + sandbox: &services.run.sandbox, tool_hooks, - services.run.cancel_token(), - ) + cancel_token: services.run.cancel_token(), + }) .await; match result { Ok(CodergenResult::Full(outcome)) => return Ok(outcome), @@ -429,7 +429,6 @@ mod tests { use tempfile::TempDir; use super::*; - use crate::event::Emitter; fn make_services() -> EngineServices { EngineServices::test_default() @@ -641,25 +640,13 @@ mod tests { #[tokio::test] async fn codergen_handler_prefers_response_text_over_status_json() { - use std::sync::Arc; - // Backend returns response text with routing directives — status.json // in the sandbox should be ignored. struct DirectiveBackend; #[async_trait] impl CodergenBackend for DirectiveBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { Ok(CodergenResult::Text { text: r#"Done. {"outcome": "succeeded", "preferred_next_label": "approve"}"# @@ -704,23 +691,11 @@ mod tests { #[tokio::test] async fn codergen_handler_extracts_status_from_last_file_touched() { - use std::sync::Arc; - struct LastFileBackend; #[async_trait] impl CodergenBackend for LastFileBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { Ok(CodergenResult::Text { text: "Done writing results.".to_string(), usage: None, @@ -772,21 +747,11 @@ mod tests { #[async_trait] impl CodergenBackend for ProviderEventBackend { - async fn run( - &self, - node: &Node, - _prompt: &str, - context: &Context, - _thread_id: Option<&str>, - emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { - let scope = StageScope::for_handler(context, &node.id); - emitter.emit_scoped( + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + let scope = StageScope::for_handler(request.context, &request.node.id); + request.emitter.emit_scoped( &crate::event::Event::AgentSessionActivated { - node_id: node.id.clone(), + node_id: request.node.id.clone(), visit: scope.visit, session_id: "session_123".to_string(), thread_id: None, @@ -883,18 +848,9 @@ mod tests { #[async_trait] impl CodergenBackend for ThreadCapturingBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { - *self.captured_thread_id.lock().unwrap() = Some(thread_id.map(String::from)); + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + *self.captured_thread_id.lock().unwrap() = + Some(request.thread_id.map(String::from)); Ok(CodergenResult::Text { text: "ok".to_string(), usage: None, @@ -936,18 +892,9 @@ mod tests { #[async_trait] impl CodergenBackend for ThreadCapturingBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { - *self.captured_thread_id.lock().unwrap() = Some(thread_id.map(String::from)); + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + *self.captured_thread_id.lock().unwrap() = + Some(request.thread_id.map(String::from)); Ok(CodergenResult::Text { text: "ok".to_string(), usage: None, @@ -984,17 +931,7 @@ mod tests { #[async_trait] impl CodergenBackend for FailingBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { Err(Error::handler("Request timed out".to_string())) } } @@ -1132,17 +1069,7 @@ Some text in between. #[async_trait] impl CodergenBackend for ValidationFailBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { Err(Error::Validation("bad config".to_string())) } } @@ -1173,18 +1100,8 @@ Some text in between. #[async_trait] impl CodergenBackend for PromptCapturingBackend { - async fn run( - &self, - _node: &Node, - prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { - *self.captured_prompt.lock().unwrap() = Some(prompt.to_string()); + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + *self.captured_prompt.lock().unwrap() = Some(request.prompt.to_string()); Ok(CodergenResult::Text { text: "ok".to_string(), usage: None, @@ -1243,18 +1160,8 @@ Some text in between. #[async_trait] impl CodergenBackend for PromptCapturingBackend { - async fn run( - &self, - _node: &Node, - prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { - *self.captured_prompt.lock().unwrap() = Some(prompt.to_string()); + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + *self.captured_prompt.lock().unwrap() = Some(request.prompt.to_string()); Ok(CodergenResult::Text { text: "ok".to_string(), usage: None, diff --git a/lib/crates/fabro-workflow/src/handler/fan_in.rs b/lib/crates/fabro-workflow/src/handler/fan_in.rs index 11a79f7b0..455b27b26 100644 --- a/lib/crates/fabro-workflow/src/handler/fan_in.rs +++ b/lib/crates/fabro-workflow/src/handler/fan_in.rs @@ -6,7 +6,7 @@ use fabro_agent::Sandbox; use fabro_graphviz::graph::{Graph, Node}; use tokio_util::sync::CancellationToken; -use super::agent::{CodergenBackend, CodergenResult}; +use super::agent::{CodergenBackend, CodergenResult, CodergenRunRequest}; use super::{EngineServices, Handler}; use crate::context::{Context, keys}; use crate::error::Error; @@ -260,16 +260,16 @@ async fn llm_evaluate( // Fan-in evaluation runs outside a thread context, so pass None match backend - .run( - &eval_node, - &full_prompt, + .run(CodergenRunRequest { + node: &eval_node, + prompt: &full_prompt, context, - None, + thread_id: None, emitter, sandbox, - None, + tool_hooks: None, cancel_token, - ) + }) .await { Ok(CodergenResult::Full(outcome)) => { @@ -469,23 +469,13 @@ mod tests { async fn fan_in_with_backend_llm_eval() { use tempfile::TempDir; - use crate::handler::agent::CodergenBackend; + use crate::handler::agent::{CodergenBackend, CodergenRunRequest}; struct MockBackend; #[async_trait] impl CodergenBackend for MockBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { // Return text that contains the ID "branch_b" Ok(CodergenResult::Text { text: "The best candidate is branch_b".to_string(), diff --git a/lib/crates/fabro-workflow/src/handler/llm/acp.rs b/lib/crates/fabro-workflow/src/handler/llm/acp.rs new file mode 100644 index 000000000..971a4c8ed --- /dev/null +++ b/lib/crates/fabro-workflow/src/handler/llm/acp.rs @@ -0,0 +1,638 @@ +//! Workflow adapter for ACP-backed LLM stages. + +use std::collections::HashMap; +use std::sync::Arc; + +use async_trait::async_trait; +use fabro_acp::{ + AcpCommandError, AcpError, AcpRunRequest, render_stop_reason, resolve_acp_command, +}; +use fabro_agent::{Sandbox, StaticEnvProvider, ToolEnvProvider}; +use fabro_auth::CredentialResolver; +use fabro_graphviz::graph::Node; +use fabro_model::Provider; +use fabro_util::time::elapsed_ms; +use tokio_util::sync::CancellationToken; + +use super::super::agent::{CodergenBackend, CodergenResult, CodergenRunRequest, OneShotRequest}; +use super::changed_files; +use super::cli::AgentCli; +use super::launch_env::{AgentLaunchEnvRequest, resolve_agent_launch_env}; +use crate::error::Error; +use crate::event::{Emitter, Event, StageScope}; + +pub struct AgentAcpBackend { + model: String, + provider: Provider, + tool_env: Option>, + github_token_refresh_managed: bool, + resolver: Option, +} + +impl AgentAcpBackend { + #[must_use] + pub fn new(model: String, provider: Provider, resolver: CredentialResolver) -> Self { + Self { + model, + provider, + tool_env: None, + github_token_refresh_managed: false, + resolver: Some(resolver), + } + } + + #[must_use] + pub fn new_from_env(model: String, provider: Provider) -> Self { + Self { + model, + provider, + tool_env: None, + github_token_refresh_managed: false, + resolver: None, + } + } + + #[must_use] + pub fn with_env(mut self, env: HashMap) -> Self { + self.tool_env = Some(Arc::new(StaticEnvProvider(env))); + self + } + + #[must_use] + pub fn with_tool_env_provider( + mut self, + provider: Arc, + github_token_refresh_managed: bool, + ) -> Self { + self.tool_env = Some(provider); + self.github_token_refresh_managed = github_token_refresh_managed; + self + } + + async fn run_turn( + &self, + node: &Node, + prompt: String, + emitter: &Arc, + stage_scope: &StageScope, + sandbox: &Arc, + cancel_token: CancellationToken, + ) -> Result { + let files_before = changed_files::detect_changed_files(sandbox).await; + let model = node.model().unwrap_or(&self.model); + let provider = node + .provider() + .and_then(|value| value.parse::().ok()) + .unwrap_or(self.provider); + let command = + resolve_acp_command(node.acp_command()).map_err(acp_command_error_to_workflow)?; + + let launch_env = resolve_agent_launch_env(AgentLaunchEnvRequest { + provider, + cli: AgentCli::for_provider(provider), + resolver: self.resolver.as_ref(), + tool_env: self.tool_env.as_ref(), + github_token_refresh_managed: self.github_token_refresh_managed, + stage_label: "ACP", + emitter, + sandbox, + cancel_token: &cancel_token, + }) + .await?; + let on_activity = { + let emitter = Arc::clone(emitter); + Arc::new(move || emitter.touch()) as Arc + }; + + let command_display = command.to_string(); + emitter.emit_scoped( + &Event::AgentAcpStarted { + node_id: node.id.clone(), + visit: stage_scope.visit, + mode: "acp".to_string(), + provider: provider.to_string(), + model: model.to_string(), + command: command_display, + }, + stage_scope, + ); + + let launch_start = std::time::Instant::now(); + let result = match fabro_acp::run_acp_turn(AcpRunRequest { + command, + prompt, + cwd: sandbox.working_directory().to_string(), + timeout_ms: node.timeout().map(crate::millis_u64), + env: launch_env, + sandbox: Arc::clone(sandbox), + cancel_token: cancel_token.child_token(), + on_activity: Some(on_activity), + }) + .await + { + Ok(result) => { + emitter.emit_scoped( + &Event::AgentAcpCompleted { + node_id: node.id.clone(), + stdout: result.text.clone(), + stderr: result.stderr.clone(), + stop_reason: render_stop_reason(&result.stop_reason), + duration_ms: result.duration_ms, + }, + stage_scope, + ); + result + } + Err(AcpError::Cancelled) => { + emitter.emit_scoped( + &Event::AgentAcpCancelled { + node_id: node.id.clone(), + stdout: String::new(), + stderr: String::new(), + duration_ms: elapsed_ms(launch_start), + }, + stage_scope, + ); + return Err(Error::Cancelled); + } + Err(AcpError::TimedOut { stderr }) => { + emitter.emit_scoped( + &Event::AgentAcpTimedOut { + node_id: node.id.clone(), + stdout: String::new(), + stderr: stderr.clone(), + duration_ms: elapsed_ms(launch_start), + }, + stage_scope, + ); + return Err(acp_error_to_workflow(AcpError::TimedOut { stderr })); + } + Err(AcpError::StopReason { stop_reason, text }) => { + emitter.emit_scoped( + &Event::AgentAcpCompleted { + node_id: node.id.clone(), + stdout: text.clone(), + stderr: String::new(), + stop_reason: stop_reason.clone(), + duration_ms: elapsed_ms(launch_start), + }, + stage_scope, + ); + return Err(acp_error_to_workflow(AcpError::StopReason { + stop_reason, + text, + })); + } + Err(error) => return Err(acp_error_to_workflow(error)), + }; + + let (files_touched, last_file_touched) = + changed_files::files_touched_since(sandbox, &files_before).await; + + Ok(CodergenResult::Text { + text: result.text, + usage: None, + files_touched, + last_file_touched, + }) + } +} + +#[async_trait] +impl CodergenBackend for AgentAcpBackend { + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + let stage_scope = StageScope::for_handler(request.context, &request.node.id); + self.run_turn( + request.node, + request.prompt.to_string(), + request.emitter, + &stage_scope, + request.sandbox, + request.cancel_token, + ) + .await + } + + async fn one_shot(&self, request: OneShotRequest<'_>) -> Result { + let prompt = match request.system_prompt.filter(|prompt| !prompt.is_empty()) { + Some(system_prompt) => format!("System:\n{system_prompt}\n\nUser:\n{}", request.prompt), + None => request.prompt.to_string(), + }; + self.run_turn( + request.node, + prompt, + request.emitter, + request.stage_scope, + request.sandbox, + request.cancel_token, + ) + .await + } +} + +fn acp_command_error_to_workflow(error: AcpCommandError) -> Error { + match error { + AcpCommandError::EmptyOverride => Error::handler("acp_command must not be empty"), + AcpCommandError::MissingOverride => Error::handler( + "acp_command is required for backend=\"acp\" because Fabro does not install ACP agents", + ), + AcpCommandError::UnsupportedTransport => { + Error::handler("only stdio ACP commands are supported") + } + AcpCommandError::Parse(source) => { + Error::handler_with_source("Failed to resolve ACP command", &source) + } + } +} + +fn acp_error_to_workflow(error: AcpError) -> Error { + match error { + AcpError::Cancelled => Error::Cancelled, + AcpError::TimedOut { stderr } => { + if stderr.is_empty() { + Error::handler("ACP turn timed out") + } else { + Error::handler(format!("ACP turn timed out: {stderr}")) + } + } + AcpError::StopReason { stop_reason, text } => { + Error::handler(format!("ACP prompt stopped with {stop_reason}: {text}")) + } + AcpError::Sandbox(source) => Error::handler_with_source("ACP turn failed", &source), + other => Error::handler_with_source("ACP turn failed", &other), + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + use std::sync::{Arc, Mutex}; + + use fabro_acp::test_support::fake_acp_agent_script; + use fabro_agent::{LocalSandbox, Sandbox, shell_quote}; + use fabro_graphviz::graph::{AttrValue, Node}; + use fabro_model::Provider; + use fabro_sandbox::test_support::MockSandbox; + use fabro_types::EventBody; + use tokio_util::sync::CancellationToken; + + use super::AgentAcpBackend; + use crate::context::Context; + use crate::event::{Emitter, StageScope}; + use crate::handler::agent::{ + CodergenBackend, CodergenResult, CodergenRunRequest, OneShotRequest, + }; + + #[tokio::test] + async fn acp_backend_run_sends_prompt_and_returns_text() { + let tempdir = tempfile::tempdir().unwrap(); + init_git(tempdir.path()); + let script_path = tempdir.path().join("fake_acp_agent.py"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + + let mut node = Node::new("work"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs.insert( + "model".to_string(), + AttrValue::String("fake-acp".to_string()), + ); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + shell_quote(&script_path.to_string_lossy()) + )), + ); + + let backend = + AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi).with_env( + HashMap::from([("ACP_MODE".to_string(), "write_file".to_string())]), + ); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + let emitter = Arc::new(Emitter::default()); + let context = Context::new(); + let result = backend + .run(CodergenRunRequest { + node: &node, + prompt: "write hello", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + let CodergenResult::Text { + text, + files_touched, + .. + } = result + else { + panic!("expected text result"); + }; + assert_eq!(text, "hello from acp"); + assert_eq!(files_touched, vec!["hello.txt"]); + } + + #[tokio::test] + async fn acp_backend_one_shot_combines_system_prompt_and_uses_passed_sandbox() { + let tempdir = tempfile::tempdir().unwrap(); + let script_path = tempdir.path().join("fake_acp_agent.py"); + let prompt_record_path = tempdir.path().join("prompt.json"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + + let mut node = Node::new("prompt"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + shell_quote(&script_path.to_string_lossy()) + )), + ); + + let backend = AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi) + .with_env(HashMap::from([ + ( + "ACP_PROMPT_RECORD".to_string(), + prompt_record_path.to_string_lossy().into_owned(), + ), + ("ACP_MODE".to_string(), "write_file".to_string()), + ])); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + let emitter = Arc::new(Emitter::default()); + let context = Context::new(); + let stage_scope = StageScope::for_handler(&context, "prompt"); + let result = backend + .one_shot(OneShotRequest { + node: &node, + prompt: "User prompt", + system_prompt: Some("System prompt"), + emitter: &emitter, + stage_scope: &stage_scope, + sandbox: &sandbox, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + assert!(matches!(result, CodergenResult::Text { .. })); + let recorded = tokio::fs::read_to_string(prompt_record_path).await.unwrap(); + assert!(recorded.contains("System:\\nSystem prompt\\n\\nUser:\\nUser prompt")); + assert_eq!( + tokio::fs::read_to_string(tempdir.path().join("hello.txt")) + .await + .unwrap(), + "hello from sandbox\n" + ); + } + + #[tokio::test] + async fn acp_backend_cancelled_stop_reason_maps_to_cancelled_error() { + let tempdir = tempfile::tempdir().unwrap(); + let script_path = tempdir.path().join("fake_acp_agent.py"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + + let mut node = Node::new("work"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + shell_quote(&script_path.to_string_lossy()) + )), + ); + + let backend = + AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi).with_env( + HashMap::from([("ACP_STOP_REASON".to_string(), "cancelled".to_string())]), + ); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + let emitter = Arc::new(Emitter::default()); + let context = Context::new(); + let result = backend + .run(CodergenRunRequest { + node: &node, + prompt: "cancel", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) + .await; + let Err(err) = result else { + panic!("expected cancellation error"); + }; + + assert!(matches!(err, crate::error::Error::Cancelled)); + } + + #[tokio::test] + async fn acp_started_event_omits_json_command_env_values() { + let tempdir = tempfile::tempdir().unwrap(); + let script_path = tempdir.path().join("fake_acp_agent.py"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + + let raw_command = serde_json::json!({ + "type": "stdio", + "name": "fake", + "command": "python3", + "args": [script_path.to_string_lossy()], + "env": [ + {"name": "OPENAI_API_KEY", "value": "secret-key"} + ], + }) + .to_string(); + let mut node = Node::new("work"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs + .insert("acp_command".to_string(), AttrValue::String(raw_command)); + + let backend = AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + let emitter = Arc::new(Emitter::default()); + let events = Arc::new(Mutex::new(Vec::new())); + emitter.on_event({ + let events = Arc::clone(&events); + move |event| events.lock().unwrap().push(event.clone()) + }); + + let context = Context::new(); + backend + .run(CodergenRunRequest { + node: &node, + prompt: "write hello", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + let events = events.lock().unwrap(); + let command = events + .iter() + .find_map(|event| match &event.body { + EventBody::AgentAcpStarted(props) => Some(props.command.as_str()), + _ => None, + }) + .expect("ACP started event should be emitted"); + assert!(command.contains("python3")); + assert!(command.contains("fake_acp_agent.py")); + assert!(!command.contains("OPENAI_API_KEY")); + assert!(!command.contains("secret-key")); + } + + #[tokio::test] + async fn acp_backend_requires_explicit_acp_command() { + let sandbox = MockSandbox::linux(); + let sandbox = Arc::new(sandbox); + let sandbox_dyn: Arc = sandbox.clone(); + + let mut node = Node::new("work"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + + let backend = AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi); + let emitter = Arc::new(Emitter::default()); + let context = Context::new(); + let result = backend + .run(CodergenRunRequest { + node: &node, + prompt: "write hello", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox_dyn, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) + .await; + let Err(err) = result else { + panic!("ACP without acp_command should fail"); + }; + assert!( + err.to_string() + .contains("acp_command is required for backend=\"acp\"") + ); + assert!( + sandbox + .captured_env_vars + .lock() + .expect("captured env lock poisoned") + .is_none(), + "ACP process should not launch when acp_command is missing" + ); + } + + #[tokio::test] + async fn acp_backend_stdio_spawn_failure_preserves_sandbox_cause() { + const DAYTONA_UNSUPPORTED_ACP: &str = "ACP backend requires bidirectional stdio; the Daytona sandbox provider does not support it yet"; + + let mut sandbox = MockSandbox::linux(); + sandbox.stdio_process_error = Some(DAYTONA_UNSUPPORTED_ACP.to_string()); + let sandbox = Arc::new(sandbox); + let sandbox_dyn: Arc = sandbox.clone(); + + let mut node = Node::new("work"); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String("fake-acp-agent".to_string()), + ); + + let backend = + AgentAcpBackend::new_from_env("fake-acp".to_string(), Provider::OpenAi).with_env( + HashMap::from([("OPENAI_API_KEY".to_string(), "test-key".to_string())]), + ); + let emitter = Arc::new(Emitter::default()); + let context = Context::new(); + let result = backend + .run(CodergenRunRequest { + node: &node, + prompt: "write hello", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox_dyn, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) + .await; + let Err(err) = result else { + panic!("stdio spawn failure should fail the ACP turn"); + }; + + let rendered = err.display_with_causes(); + assert!( + rendered.contains("ACP turn failed"), + "rendered error should keep ACP context: {rendered}" + ); + assert!( + err.causes() + .iter() + .any(|cause| cause == DAYTONA_UNSUPPORTED_ACP), + "cause chain should include sandbox failure, got: {rendered}" + ); + assert_eq!( + err.failure_category(), + crate::error::FailureCategory::Deterministic + ); + } + + #[expect( + clippy::disallowed_methods, + reason = "unit test initializes an isolated git repository with the system git binary" + )] + fn init_git(path: &std::path::Path) { + let output = std::process::Command::new("git") + .arg("init") + .current_dir(path) + .output() + .unwrap(); + assert!(output.status.success()); + } +} diff --git a/lib/crates/fabro-workflow/src/handler/llm/api.rs b/lib/crates/fabro-workflow/src/handler/llm/api.rs index a65093d85..a8cdfab03 100644 --- a/lib/crates/fabro-workflow/src/handler/llm/api.rs +++ b/lib/crates/fabro-workflow/src/handler/llm/api.rs @@ -19,10 +19,10 @@ use tokio::sync::Mutex as TokioMutex; use tokio::task::JoinHandle; use tokio_util::sync::CancellationToken; -use super::super::agent::{CodergenBackend, CodergenResult}; +use super::super::agent::{CodergenBackend, CodergenResult, CodergenRunRequest, OneShotRequest}; use super::activation_lease::{ActivationLease, ActivationLeaseOptions}; +use crate::context::WorkflowContext; use crate::context::keys::Fidelity; -use crate::context::{Context, WorkflowContext}; use crate::error::Error; use crate::event::{Emitter, Event, StageScope}; use crate::outcome::billed_model_usage_from_llm; @@ -476,14 +476,13 @@ impl CodergenBackend for AgentApiBackend { self.shutdown_cached_sessions(emitter); } - async fn one_shot( - &self, - node: &Node, - prompt: &str, - system_prompt: Option<&str>, - emitter: &Arc, - stage_scope: &StageScope, - ) -> Result { + async fn one_shot(&self, request: OneShotRequest<'_>) -> Result { + let node = request.node; + let prompt = request.prompt; + let system_prompt = request.system_prompt; + let emitter = request.emitter; + let stage_scope = request.stage_scope; + let client = Client::from_source(self.source.as_ref()) .await .map_err(|e| Error::handler_with_source("Failed to create LLM client", &e))?; @@ -617,17 +616,16 @@ impl CodergenBackend for AgentApiBackend { }) } - async fn run( - &self, - node: &Node, - prompt: &str, - context: &Context, - thread_id: Option<&str>, - emitter: &Arc, - sandbox: &Arc, - tool_hooks: Option>, - cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + let node = request.node; + let prompt = request.prompt; + let context = request.context; + let thread_id = request.thread_id; + let emitter = request.emitter; + let sandbox = request.sandbox; + let tool_hooks = request.tool_hooks; + let cancel_token = request.cancel_token; + let actual_model = node.model().unwrap_or(&self.model).to_string(); let _actual_provider = node .provider() diff --git a/lib/crates/fabro-workflow/src/handler/llm/changed_files.rs b/lib/crates/fabro-workflow/src/handler/llm/changed_files.rs new file mode 100644 index 000000000..6b07e2f5d --- /dev/null +++ b/lib/crates/fabro-workflow/src/handler/llm/changed_files.rs @@ -0,0 +1,83 @@ +use std::collections::HashSet; +use std::sync::Arc; + +use fabro_agent::{Sandbox, shell_quote}; + +const DIFF_MARKER: &str = "__FABRO_CHANGED_FILES_DIFF__"; +const UNTRACKED_MARKER: &str = "__FABRO_CHANGED_FILES_UNTRACKED__"; + +pub async fn detect_changed_files(sandbox: &Arc) -> Vec { + let mut files: Vec = Vec::new(); + let command = format!( + "printf '%s\\n' {diff}; git diff --name-only || true; \ + printf '%s\\n' {untracked}; git ls-files --others --exclude-standard || true", + diff = shell_quote(DIFF_MARKER), + untracked = shell_quote(UNTRACKED_MARKER), + ); + if let Ok(result) = sandbox + .exec_command(&command, 30_000, None, None, None) + .await + { + if result.is_success() { + files.extend(parse_changed_files(&result.stdout)); + } + } + + files.sort(); + files.dedup(); + files +} + +pub async fn files_touched_since( + sandbox: &Arc, + files_before: &[String], +) -> (Vec, Option) { + let files_after = detect_changed_files(sandbox).await; + let files_before: HashSet<&str> = files_before.iter().map(String::as_str).collect(); + let files_touched: Vec = files_after + .into_iter() + .filter(|file| !files_before.contains(file.as_str())) + .collect(); + + let last_file_touched = if files_touched.is_empty() { + None + } else { + let quoted_files: Vec = + files_touched.iter().map(|file| shell_quote(file)).collect(); + let cmd = format!("ls -t {} | head -1", quoted_files.join(" ")); + sandbox + .exec_command(&cmd, 5_000, None, None, None) + .await + .ok() + .and_then(|result| { + let trimmed = result.stdout.trim().to_string(); + (result.is_success() && !trimmed.is_empty()).then_some(trimmed) + }) + }; + + (files_touched, last_file_touched) +} + +fn parse_changed_files(stdout: &str) -> impl Iterator + '_ { + stdout.lines().filter_map(|line| { + let trimmed = line.trim(); + (!trimmed.is_empty() && trimmed != DIFF_MARKER && trimmed != UNTRACKED_MARKER) + .then(|| trimmed.to_string()) + }) +} + +#[cfg(test)] +mod tests { + use super::parse_changed_files; + + #[test] + fn parse_changed_files_ignores_section_markers() { + let files = parse_changed_files( + "__FABRO_CHANGED_FILES_DIFF__\nsrc/main.rs\n\ + __FABRO_CHANGED_FILES_UNTRACKED__\nREADME.md\n", + ) + .collect::>(); + + assert_eq!(files, vec!["src/main.rs", "README.md"]); + } +} diff --git a/lib/crates/fabro-workflow/src/handler/llm/cli.rs b/lib/crates/fabro-workflow/src/handler/llm/cli.rs index c69f8d99c..129583c83 100644 --- a/lib/crates/fabro-workflow/src/handler/llm/cli.rs +++ b/lib/crates/fabro-workflow/src/handler/llm/cli.rs @@ -8,11 +8,11 @@ use std::sync::{Arc, Mutex}; use async_trait::async_trait; use fabro_agent::{Sandbox, StaticEnvProvider, ToolEnvProvider, shell_quote}; -use fabro_auth::{CliAgentKind, CredentialResolver, CredentialUsage, ResolvedCredential}; +use fabro_auth::CredentialResolver; use fabro_graphviz::graph::Node; use fabro_llm::types::TokenCounts; use fabro_model::Provider; -use fabro_types::{CommandOutputStream, CommandTermination}; +use fabro_types::{CommandOutputStream, CommandTermination, LlmBackend}; use fabro_util::time::elapsed_ms; use tokio_util::sync::CancellationToken; @@ -38,10 +38,12 @@ fn cli_failure_detail(stdout: &str, stderr: &str, command: &str) -> String { } } -use super::super::agent::{CodergenBackend, CodergenResult}; -use crate::context::Context; +use super::super::agent::{CodergenBackend, CodergenResult, CodergenRunRequest, OneShotRequest}; +use super::acp::AgentAcpBackend; +use super::launch_env::{AgentLaunchEnvRequest, resolve_agent_launch_env}; +use super::{changed_files, routing}; use crate::error::Error; -use crate::event::{Emitter, Event, RunNoticeCode, RunNoticeLevel, StageScope}; +use crate::event::{Emitter, Event, StageScope}; use crate::outcome::billed_model_usage_from_llm; /// Maps a provider to its corresponding CLI tool metadata. @@ -73,41 +75,20 @@ impl AgentCli { Self::Gemini => "gemini", } } - - pub fn npm_package(self) -> &'static str { - match self { - Self::Claude => "@anthropic-ai/claude-code", - Self::Codex => "@openai/codex", - Self::Gemini => "@anthropic-ai/gemini-cli", - } - } } -/// Ensure the CLI tool for the given provider is installed in the sandbox. -/// -/// Checks if the CLI binary exists; if not, installs Node.js (if missing) and -/// the CLI via npm. Emits `CliEnsure*` events for observability. -async fn ensure_cli( +/// Verify the provider CLI exists in the sandbox. Fabro does not install agent +/// CLIs at runtime; sandbox images or setup steps own tool installation. +async fn verify_cli_available( cli: AgentCli, - provider: Provider, sandbox: &Arc, - emitter: &Arc, cancel_token: &CancellationToken, ) -> Result<(), Error> { - let start = std::time::Instant::now(); let cli_name = cli.name(); - let provider_str = <&'static str>::from(provider); - emitter.emit(&Event::CliEnsureStarted { - cli_name: cli_name.to_string(), - provider: provider_str.to_string(), - }); - - // Check if the CLI is already installed (include ~/.local/bin for npm-installed - // CLIs) - let version_check = sandbox + let availability_check = sandbox .exec_command( - &format!("PATH=\"$HOME/.local/bin:$PATH\" {cli_name} --version"), + &format!("PATH=\"$HOME/.local/bin:$PATH\" command -v {cli_name}"), 30_000, None, None, @@ -115,68 +96,17 @@ async fn ensure_cli( ) .await .map_err(|e| { - Error::handler_with_source(format!("Failed to check {cli_name} version"), &e) + Error::handler_with_source(format!("Failed to check {cli_name} availability"), &e) })?; - if version_check.is_success() { - let duration_ms = elapsed_ms(start); - emitter.emit(&Event::CliEnsureCompleted { - cli_name: cli_name.to_string(), - provider: provider_str.to_string(), - already_installed: true, - node_installed: false, - duration_ms, - }); + if availability_check.is_success() { return Ok(()); } - // Install Node.js (if needed) and the CLI in a single shell so PATH persists - let install_cmd = format!( - "export PATH=\"$HOME/.local/bin:$PATH\" && \ - (node --version >/dev/null 2>&1 || \ - (mkdir -p ~/.local && curl -fsSL https://nodejs.org/dist/v22.14.0/node-v22.14.0-linux-x64.tar.gz | tar -xz --strip-components=1 -C ~/.local)) && \ - npm install -g {}", - cli.npm_package() - ); - let install_result = sandbox - .exec_command( - &install_cmd, - 180_000, - None, - None, - Some(cancel_token.child_token()), - ) - .await - .map_err(|e| Error::handler_with_source(format!("Failed to install {cli_name}"), &e))?; - - let node_installed = true; - if !install_result.is_success() { - let duration_ms = elapsed_ms(start); - let exec_output_tail = install_result.default_redacted_output_tail(); - let error_msg = format!( - "{cli_name} install exited with code {}", - install_result.display_exit_code() - ); - emitter.emit(&Event::CliEnsureFailed { - cli_name: cli_name.to_string(), - provider: provider_str.to_string(), - error: error_msg.clone(), - duration_ms, - exec_output_tail, - }); - return Err(Error::handler(error_msg)); - } - - let duration_ms = elapsed_ms(start); - emitter.emit(&Event::CliEnsureCompleted { - cli_name: cli_name.to_string(), - provider: provider_str.to_string(), - already_installed: false, - node_installed, - duration_ms, - }); - - Ok(()) + Err(Error::handler(format!( + "CLI backend requires '{cli_name}' to be installed in the sandbox PATH. Install it in the \ + sandbox image or setup steps before running backend=\"cli\"." + ))) } /// Models that are only available through CLI tools (not via API). @@ -404,7 +334,6 @@ pub struct AgentCliBackend { provider: Provider, tool_env: Option>, github_token_refresh_managed: bool, - poll_interval: std::time::Duration, resolver: Option, } @@ -416,7 +345,6 @@ impl AgentCliBackend { provider, tool_env: None, github_token_refresh_managed: false, - poll_interval: std::time::Duration::from_secs(5), resolver: Some(resolver), } } @@ -428,7 +356,6 @@ impl AgentCliBackend { provider, tool_env: None, github_token_refresh_managed: false, - poll_interval: std::time::Duration::from_secs(5), resolver: None, } } @@ -449,79 +376,20 @@ impl AgentCliBackend { self.github_token_refresh_managed = github_token_refresh_managed; self } - - #[must_use] - pub fn with_poll_interval(mut self, interval: std::time::Duration) -> Self { - self.poll_interval = interval; - self - } - - /// Detect changed files by comparing git state before and after the CLI - /// run. - async fn detect_changed_files(&self, sandbox: &Arc) -> Vec { - // Get unstaged changes - let diff_result = sandbox - .exec_command("git diff --name-only", 30_000, None, None, None) - .await; - - // Get untracked files - let untracked_result = sandbox - .exec_command( - "git ls-files --others --exclude-standard", - 30_000, - None, - None, - None, - ) - .await; - - let mut files: Vec = Vec::new(); - - if let Ok(result) = diff_result { - if result.is_success() { - files.extend( - result - .stdout - .lines() - .filter(|l| !l.trim().is_empty()) - .map(String::from), - ); - } - } - - if let Ok(result) = untracked_result { - if result.is_success() { - files.extend( - result - .stdout - .lines() - .filter(|l| !l.trim().is_empty()) - .map(String::from), - ); - } - } - - files.sort(); - files.dedup(); - files - } } #[async_trait] impl CodergenBackend for AgentCliBackend { - async fn run( - &self, - node: &Node, - prompt: &str, - context: &Context, - _thread_id: Option<&str>, - emitter: &Arc, - sandbox: &Arc, - _tool_hooks: Option>, - cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + let node = request.node; + let prompt = request.prompt; + let context = request.context; + let emitter = request.emitter; + let sandbox = request.sandbox; + let cancel_token = request.cancel_token; + // 1. Snapshot git state before the CLI run - let files_before = self.detect_changed_files(sandbox).await; + let files_before = changed_files::detect_changed_files(sandbox).await; // 2. Generate unique paths for this run let run_id = uuid::Uuid::new_v4().to_string(); @@ -541,9 +409,8 @@ impl CodergenBackend for AgentCliBackend { .and_then(|s| s.parse::().ok()) .unwrap_or(self.provider); - // Ensure the CLI tool is installed in the sandbox let cli = AgentCli::for_provider(provider); - ensure_cli(cli, provider, sandbox, emitter, &cancel_token).await?; + verify_cli_available(cli, sandbox, &cancel_token).await?; let command = cli_command_for_provider(provider, model, &prompt_path); let stage_scope = StageScope::for_handler(context, &node.id); @@ -559,67 +426,18 @@ impl CodergenBackend for AgentCliBackend { &stage_scope, ); - // Forward provider API key and custom env vars so the CLI tool can - // authenticate. Resolve credentials and run any pre-login command - // before the main CLI invocation. - let cli_agent = match cli { - AgentCli::Claude => CliAgentKind::Claude, - AgentCli::Codex => CliAgentKind::Codex, - AgentCli::Gemini => CliAgentKind::Gemini, - }; - let mut launch_env = if let Some(resolver) = &self.resolver { - let resolved = resolver - .resolve(provider, CredentialUsage::CliAgent(cli_agent)) - .await - .map_err(|e| Error::handler_with_source("Failed to resolve CLI credential", &e))?; - let ResolvedCredential::Cli(cli_credential) = resolved else { - return Err(Error::handler("Expected CLI credential".to_string())); - }; - if let Some(login_cmd) = &cli_credential.login_command { - let login_result = sandbox - .exec_command( - login_cmd, - 30_000, - None, - None, - Some(cancel_token.child_token()), - ) - .await - .map_err(|e| Error::handler_with_source("codex login failed", &e))?; - if !login_result.is_success() { - tracing::warn!( - exit_code = login_result.display_exit_code(), - "codex login --with-api-key failed: {}", - login_result.stderr - ); - } - } - cli_credential.env_vars - } else { - let mut env = HashMap::new(); - for name in provider.api_key_env_vars() { - if let Some(val) = process_env_var(name) { - env.insert((*name).to_string(), val); - } - } - env - }; - if let Some(provider) = &self.tool_env { - if self.github_token_refresh_managed { - emitter.notice( - RunNoticeLevel::Info, - RunNoticeCode::GithubTokenRefreshLimited, - "CLI agent stages receive GitHub tokens at process launch; stages running \ - beyond token expiry may need to be retried.", - ); - } - let tool_env = provider.resolve().await.map_err(|err| { - Error::handler_with_anyhow("Failed to resolve CLI agent env", &err) - })?; - for (name, val) in tool_env { - launch_env.insert(name, val); - } - } + let launch_env = resolve_agent_launch_env(AgentLaunchEnvRequest { + provider, + cli, + resolver: self.resolver.as_ref(), + tool_env: self.tool_env.as_ref(), + github_token_refresh_managed: self.github_token_refresh_managed, + stage_label: "CLI", + emitter, + sandbox, + cancel_token: &cancel_token, + }) + .await?; // Write env file so the inner shell that runs the CLI command picks up // PATH and provider env vars; we still pass `launch_env` to @@ -801,29 +619,8 @@ impl CodergenBackend for AgentCliBackend { .ok_or_else(|| Error::handler("Failed to parse CLI output".to_string()))?; // 5. Detect changed files - let files_after = self.detect_changed_files(sandbox).await; - let files_touched: Vec = files_after - .into_iter() - .filter(|f| !files_before.contains(f)) - .collect(); - - // Find the most recently modified file by mtime - let last_file_touched = if files_touched.is_empty() { - None - } else { - let quoted_files: Vec = files_touched.iter().map(|f| shell_quote(f)).collect(); - let cmd = format!("ls -t {} | head -1", quoted_files.join(" ")); - if let Ok(result) = sandbox.exec_command(&cmd, 5_000, None, None, None).await { - let trimmed = result.stdout.trim().to_string(); - if result.is_success() && !trimmed.is_empty() { - Some(trimmed) - } else { - None - } - } else { - None - } - }; + let (files_touched, last_file_touched) = + changed_files::files_touched_since(sandbox, &files_before).await; let stage_usage = billed_model_usage_from_llm(model, provider, node.speed(), &TokenCounts { @@ -845,105 +642,65 @@ impl CodergenBackend for AgentCliBackend { clippy::disallowed_methods, reason = "CLI agent fallback credentials intentionally read provider API-key env vars." )] -fn process_env_var(name: &str) -> Option { +pub(crate) fn process_env_var(name: &str) -> Option { std::env::var(name).ok() } -/// Routes codergen invocations to either the API backend or CLI backend -/// based on node attributes and model type. +/// Routes codergen invocations to API, CLI, or ACP backends based on node +/// attributes and model type. pub struct BackendRouter { - api_backend: Box, - cli_backend: AgentCliBackend, + api: Box, + cli: AgentCliBackend, + acp: AgentAcpBackend, } impl BackendRouter { #[must_use] - pub fn new(api_backend: Box, cli_backend: AgentCliBackend) -> Self { + pub fn new( + api_backend: Box, + cli_backend: AgentCliBackend, + acp_backend: AgentAcpBackend, + ) -> Self { Self { - api_backend, - cli_backend, + api: api_backend, + cli: cli_backend, + acp: acp_backend, } } - #[allow( - clippy::unused_self, - reason = "CLI backend selection lives on the router even though it only inspects the node." - )] - fn should_use_cli(&self, node: &Node) -> bool { - // Explicit backend="cli" attribute on the node - if node.backend() == Some("cli") { - return true; - } + fn select_backend(node: &Node) -> Result { + routing::select_run_backend(node) + } - // CLI-only model on the node - if let Some(model) = node.model() { - if is_cli_only_model(model) { - return true; - } - } + fn select_one_shot_backend(node: &Node) -> Result { + routing::select_one_shot_backend(node) + } - false + #[cfg(test)] + fn should_use_cli(node: &Node) -> bool { + matches!(Self::select_backend(node), Ok(LlmBackend::Cli)) } } #[async_trait] impl CodergenBackend for BackendRouter { - async fn run( - &self, - node: &Node, - prompt: &str, - context: &Context, - thread_id: Option<&str>, - emitter: &Arc, - sandbox: &Arc, - tool_hooks: Option>, - cancel_token: CancellationToken, - ) -> Result { - if self.should_use_cli(node) { - self.cli_backend - .run( - node, - prompt, - context, - thread_id, - emitter, - sandbox, - tool_hooks, - cancel_token, - ) - .await - } else { - self.api_backend - .run( - node, - prompt, - context, - thread_id, - emitter, - sandbox, - tool_hooks, - cancel_token, - ) - .await + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + match Self::select_backend(request.node)? { + LlmBackend::Api => self.api.run(request).await, + LlmBackend::Cli => self.cli.run(request).await, + LlmBackend::Acp => self.acp.run(request).await, } } - async fn one_shot( - &self, - node: &Node, - prompt: &str, - system_prompt: Option<&str>, - emitter: &Arc, - stage_scope: &StageScope, - ) -> Result { - // CLI backend doesn't support one_shot, always route to API - self.api_backend - .one_shot(node, prompt, system_prompt, emitter, stage_scope) - .await + async fn one_shot(&self, request: OneShotRequest<'_>) -> Result { + match Self::select_one_shot_backend(request.node)? { + LlmBackend::Acp => self.acp.one_shot(request).await, + LlmBackend::Api | LlmBackend::Cli => self.api.one_shot(request).await, + } } async fn shutdown(&self, emitter: &Arc) { - self.api_backend.shutdown(emitter).await; + self.api.shutdown(emitter).await; } } @@ -951,10 +708,12 @@ impl CodergenBackend for BackendRouter { mod tests { use std::path::Path; + use fabro_agent::LocalSandbox; use fabro_agent::sandbox::ExecResult; use fabro_graphviz::graph::AttrValue; use super::*; + use crate::context::Context; // -- AgentCli -- @@ -979,18 +738,12 @@ mod tests { assert_eq!(AgentCli::Gemini.name(), "gemini"); } - #[test] - fn agent_cli_npm_package() { - assert_eq!(AgentCli::Claude.npm_package(), "@anthropic-ai/claude-code"); - assert_eq!(AgentCli::Codex.npm_package(), "@openai/codex"); - assert_eq!(AgentCli::Gemini.npm_package(), "@anthropic-ai/gemini-cli"); - } - - // -- ensure_cli -- + // -- verify_cli_available -- use std::collections::VecDeque; use std::sync::Mutex; + use fabro_acp::test_support::fake_acp_agent_script; use fabro_agent::sandbox::{DirEntry, GrepOptions}; /// Mock sandbox that returns pre-configured ExecResults in FIFO order. @@ -1123,113 +876,47 @@ mod tests { } #[tokio::test] - async fn ensure_cli_skips_install_when_present() { + async fn verify_cli_available_succeeds_when_present() { let commands = Arc::new(Mutex::new(Vec::new())); let sandbox: Arc = Arc::new(CliMockSandbox::new( vec![ok_result()], Arc::clone(&commands), )); - let emitter = Arc::new(Emitter::default()); - - let result = ensure_cli( - AgentCli::Claude, - Provider::Anthropic, - &sandbox, - &emitter, - &CancellationToken::new(), - ) - .await; + let result = + verify_cli_available(AgentCli::Claude, &sandbox, &CancellationToken::new()).await; assert!(result.is_ok()); let commands = commands.lock().unwrap(); assert_eq!(commands.len(), 1); - assert!(commands[0].contains("claude --version")); + assert!(commands[0].contains("command -v claude")); } #[tokio::test] - async fn ensure_cli_installs_when_missing() { + async fn verify_cli_available_fails_when_missing_without_installing() { let commands = Arc::new(Mutex::new(Vec::new())); - // version check fails, combined install succeeds let sandbox: Arc = Arc::new(CliMockSandbox::new( - vec![ - fail_result(127), // claude --version - ok_result(), // combined node + npm install - ], + vec![fail_result(127)], Arc::clone(&commands), )); - let emitter = Arc::new(Emitter::default()); - let result = ensure_cli( - AgentCli::Claude, - Provider::Anthropic, - &sandbox, - &emitter, - &CancellationToken::new(), - ) - .await; - assert!(result.is_ok()); + let result = + verify_cli_available(AgentCli::Claude, &sandbox, &CancellationToken::new()).await; + assert!(result.is_err()); + assert!( + result + .unwrap_err() + .to_string() + .contains("CLI backend requires 'claude' to be installed") + ); let commands = commands.lock().unwrap(); - assert_eq!(commands.len(), 2); - assert!(commands[1].contains("npm install -g @anthropic-ai/claude-code")); - } - - #[tokio::test] - async fn ensure_cli_fails_on_install_failure() { - let commands = Arc::new(Mutex::new(Vec::new())); - let sandbox: Arc = Arc::new(CliMockSandbox::new( - vec![ - fail_result(127), // claude --version - fail_result_with_output(1, "install stdout detail", "install stderr detail"), - ], - Arc::clone(&commands), - )); - let emitter = Arc::new(Emitter::default()); - let events = Arc::new(Mutex::new(Vec::::new())); - emitter.on_event({ - let events = Arc::clone(&events); - move |event| events.lock().unwrap().push(event.clone()) - }); - - let result = ensure_cli( - AgentCli::Claude, - Provider::Anthropic, - &sandbox, - &emitter, - &CancellationToken::new(), - ) - .await; - assert!(result.is_err()); - let error = result.unwrap_err().to_string(); - assert!(error.contains("install exited with code 1")); - assert!(!error.contains("install stdout detail")); - assert!(!error.contains("install stderr detail")); - - let events = events.lock().unwrap(); - let failed = events - .iter() - .find(|event| event.event_name() == "cli.ensure.failed") - .expect("cli ensure failed event"); - match &failed.body { - fabro_types::EventBody::CliEnsureFailed(props) => { - assert_eq!(props.error, "claude install exited with code 1"); - assert_eq!( - props - .exec_output_tail - .as_ref() - .and_then(|tail| tail.stdout.as_deref()), - Some("install stdout detail") - ); - assert_eq!( - props - .exec_output_tail - .as_ref() - .and_then(|tail| tail.stderr.as_deref()), - Some("install stderr detail") - ); - } - other => panic!("expected cli ensure failed body, got {other:?}"), - } + assert_eq!(commands.len(), 1); + assert!(commands[0].contains("command -v claude")); + assert!( + !commands + .iter() + .any(|command| command.contains("npm install")) + ); } // -- Cycle 1: cli_command_for_provider -- @@ -1388,18 +1075,14 @@ mod tests { node.attrs .insert("backend".to_string(), AttrValue::String("cli".to_string())); - let cli_backend = AgentCliBackend::new_from_env("model".into(), Provider::Anthropic); - let router = BackendRouter::new(Box::new(StubBackend), cli_backend); - assert!(router.should_use_cli(&node)); + assert!(BackendRouter::should_use_cli(&node)); } #[test] fn router_uses_api_by_default() { let node = Node::new("test"); - let cli_backend = AgentCliBackend::new_from_env("model".into(), Provider::Anthropic); - let router = BackendRouter::new(Box::new(StubBackend), cli_backend); - assert!(!router.should_use_cli(&node)); + assert!(!BackendRouter::should_use_cli(&node)); } #[test] @@ -1410,9 +1093,168 @@ mod tests { AttrValue::String("claude-opus-4-6".to_string()), ); + assert!(!BackendRouter::should_use_cli(&node)); + } + + #[test] + fn router_uses_api_for_backend_api() { + let mut node = Node::new("test"); + node.attrs + .insert("backend".to_string(), AttrValue::String("api".to_string())); + + assert_eq!( + BackendRouter::select_backend(&node).unwrap(), + LlmBackend::Api + ); + } + + #[test] + fn router_uses_cli_for_backend_cli() { + let mut node = Node::new("test"); + node.attrs + .insert("backend".to_string(), AttrValue::String("cli".to_string())); + + assert_eq!( + BackendRouter::select_backend(&node).unwrap(), + LlmBackend::Cli + ); + } + + #[test] + fn router_uses_acp_for_backend_acp() { + let mut node = Node::new("test"); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + + assert_eq!( + BackendRouter::select_backend(&node).unwrap(), + LlmBackend::Acp + ); + } + + #[test] + fn router_rejects_unknown_backend() { + let mut node = Node::new("test"); + node.attrs.insert( + "backend".to_string(), + AttrValue::String("codex".to_string()), + ); + + let err = BackendRouter::select_backend(&node).unwrap_err(); + assert_eq!( + err.to_string(), + "Validation error: unsupported LLM backend \"codex\"; expected one of: api, cli, acp" + ); + } + + #[tokio::test] + async fn router_routes_one_shot_to_acp_for_backend_acp() { + let tempdir = tempfile::tempdir().unwrap(); + let script_path = tempdir.path().join("fake_acp_agent.py"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + let sandbox: Arc = Arc::new(LocalSandbox::new(tempdir.path().to_path_buf())); + let mut node = Node::new("test"); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + shell_quote(&script_path.to_string_lossy()) + )), + ); + + let context = Context::new(); + let router = test_router(); + let emitter = Arc::new(Emitter::default()); + let stage_scope = StageScope::for_handler(&context, "test"); + let result = router + .one_shot(OneShotRequest { + node: &node, + prompt: "prompt", + system_prompt: None, + emitter: &emitter, + stage_scope: &stage_scope, + sandbox: &sandbox, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + let CodergenResult::Text { text, .. } = result else { + panic!("expected text result"); + }; + assert_eq!(text, "hello from acp"); + } + + #[tokio::test] + async fn router_routes_one_shot_to_api_by_default() { + let node = Node::new("test"); + let sandbox: Arc = Arc::new(LocalSandbox::new( + tempfile::tempdir().unwrap().path().to_path_buf(), + )); + let context = Context::new(); + let router = test_router(); + let emitter = Arc::new(Emitter::default()); + let stage_scope = StageScope::for_handler(&context, "test"); + + let result = router + .one_shot(OneShotRequest { + node: &node, + prompt: "prompt", + system_prompt: None, + emitter: &emitter, + stage_scope: &stage_scope, + sandbox: &sandbox, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + let CodergenResult::Text { text, .. } = result else { + panic!("expected text result"); + }; + assert_eq!(text, "api one-shot"); + } + + #[tokio::test] + async fn router_routes_one_shot_to_api_for_legacy_cli_backend() { + let mut node = Node::new("test"); + node.attrs + .insert("backend".to_string(), AttrValue::String("cli".to_string())); + let sandbox: Arc = Arc::new(LocalSandbox::new( + tempfile::tempdir().unwrap().path().to_path_buf(), + )); + let context = Context::new(); + let router = test_router(); + let emitter = Arc::new(Emitter::default()); + let stage_scope = StageScope::for_handler(&context, "test"); + + let result = router + .one_shot(OneShotRequest { + node: &node, + prompt: "prompt", + system_prompt: None, + emitter: &emitter, + stage_scope: &stage_scope, + sandbox: &sandbox, + cancel_token: CancellationToken::new(), + }) + .await + .unwrap(); + + let CodergenResult::Text { text, .. } = result else { + panic!("expected text result"); + }; + assert_eq!(text, "api one-shot"); + } + + fn test_router() -> BackendRouter { let cli_backend = AgentCliBackend::new_from_env("model".into(), Provider::Anthropic); - let router = BackendRouter::new(Box::new(StubBackend), cli_backend); - assert!(!router.should_use_cli(&node)); + let acp_backend = AgentAcpBackend::new_from_env("model".into(), Provider::Anthropic); + BackendRouter::new(Box::new(StubBackend), cli_backend, acp_backend) } /// Minimal stub backend for testing routing logic. @@ -1420,17 +1262,7 @@ mod tests { #[async_trait] impl CodergenBackend for StubBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { Ok(CodergenResult::Text { text: "stub".to_string(), usage: None, @@ -1438,6 +1270,15 @@ mod tests { last_file_touched: None, }) } + + async fn one_shot(&self, _request: OneShotRequest<'_>) -> Result { + Ok(CodergenResult::Text { + text: "api one-shot".to_string(), + usage: None, + files_touched: Vec::new(), + last_file_touched: None, + }) + } } /// Sandbox stub whose `exec_command_streaming` returns a configurable @@ -1484,8 +1325,8 @@ mod tests { _cancel_token: Option, ) -> fabro_sandbox::Result { self.commands.lock().unwrap().push(command.to_string()); - // Default: success for git/version/cat/rm/ls. - if command.contains("--version") { + // Default: success for CLI availability checks and lightweight setup. + if command.contains("command -v ") { return Ok(ok_result()); } Ok(ExecResult { @@ -1581,16 +1422,16 @@ mod tests { let events = collect_events(&emitter); let result = backend - .run( - &node, - "Do something", - &context, - None, - &emitter, - &sandbox, - None, - CancellationToken::new(), - ) + .run(CodergenRunRequest { + node: &node, + prompt: "Do something", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) .await; let Err(err) = result else { @@ -1634,16 +1475,16 @@ mod tests { let events = collect_events(&emitter); let result = backend - .run( - &node, - "Do something slow", - &context, - None, - &emitter, - &sandbox, - None, - CancellationToken::new(), - ) + .run(CodergenRunRequest { + node: &node, + prompt: "Do something slow", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) .await; let Err(err) = result else { diff --git a/lib/crates/fabro-workflow/src/handler/llm/launch_env.rs b/lib/crates/fabro-workflow/src/handler/llm/launch_env.rs new file mode 100644 index 000000000..134dea313 --- /dev/null +++ b/lib/crates/fabro-workflow/src/handler/llm/launch_env.rs @@ -0,0 +1,107 @@ +use std::collections::HashMap; +use std::sync::Arc; + +use fabro_agent::{Sandbox, ToolEnvProvider}; +use fabro_auth::{CliAgentKind, CredentialResolver, CredentialUsage, ResolvedCredential}; +use fabro_model::Provider; +use tokio_util::sync::CancellationToken; + +use super::cli::{AgentCli, process_env_var}; +use crate::error::Error; +use crate::event::{Emitter, RunNoticeCode, RunNoticeLevel}; + +pub(crate) struct AgentLaunchEnvRequest<'a> { + pub provider: Provider, + pub cli: AgentCli, + pub resolver: Option<&'a CredentialResolver>, + pub tool_env: Option<&'a Arc>, + pub github_token_refresh_managed: bool, + pub stage_label: &'static str, + pub emitter: &'a Arc, + pub sandbox: &'a Arc, + pub cancel_token: &'a CancellationToken, +} + +pub(crate) async fn resolve_agent_launch_env( + request: AgentLaunchEnvRequest<'_>, +) -> Result, Error> { + let cli_agent = match request.cli { + AgentCli::Claude => CliAgentKind::Claude, + AgentCli::Codex => CliAgentKind::Codex, + AgentCli::Gemini => CliAgentKind::Gemini, + }; + + let mut launch_env = if let Some(resolver) = request.resolver { + let resolved = resolver + .resolve(request.provider, CredentialUsage::CliAgent(cli_agent)) + .await + .map_err(|err| { + Error::handler_with_source( + format!("Failed to resolve {} credential", request.stage_label), + &err, + ) + })?; + let ResolvedCredential::Cli(cli_credential) = resolved else { + return Err(Error::handler("Expected CLI credential".to_string())); + }; + if let Some(login_cmd) = &cli_credential.login_command { + let login_result = request + .sandbox + .exec_command( + login_cmd, + 30_000, + None, + None, + Some(request.cancel_token.child_token()), + ) + .await + .map_err(|err| { + Error::handler_with_source( + format!("{} credential login failed", request.stage_label), + &err, + ) + })?; + if !login_result.is_success() { + tracing::warn!( + exit_code = login_result.display_exit_code(), + stage = request.stage_label, + "{} credential login failed: {}", + request.stage_label, + login_result.stderr + ); + } + } + cli_credential.env_vars + } else { + let mut env = HashMap::new(); + for name in request.provider.api_key_env_vars() { + if let Some(value) = process_env_var(name) { + env.insert((*name).to_string(), value); + } + } + env + }; + + if let Some(provider) = request.tool_env { + if request.github_token_refresh_managed { + request.emitter.notice( + RunNoticeLevel::Info, + RunNoticeCode::GithubTokenRefreshLimited, + format!( + "{} agent stages receive GitHub tokens at process launch; stages running \ + beyond token expiry may need to be retried.", + request.stage_label + ), + ); + } + let tool_env = provider.resolve().await.map_err(|err| { + Error::handler_with_anyhow( + format!("Failed to resolve {} agent env", request.stage_label), + &err, + ) + })?; + launch_env.extend(tool_env); + } + + Ok(launch_env) +} diff --git a/lib/crates/fabro-workflow/src/handler/llm/mod.rs b/lib/crates/fabro-workflow/src/handler/llm/mod.rs index f210015ff..6a19ef8b5 100644 --- a/lib/crates/fabro-workflow/src/handler/llm/mod.rs +++ b/lib/crates/fabro-workflow/src/handler/llm/mod.rs @@ -1,7 +1,12 @@ +pub mod acp; pub mod activation_lease; pub mod api; +pub mod changed_files; pub mod cli; +pub mod launch_env; pub mod preamble; +pub mod routing; +pub use acp::AgentAcpBackend; pub use api::AgentApiBackend; pub use cli::{AgentCliBackend, BackendRouter, parse_cli_response}; diff --git a/lib/crates/fabro-workflow/src/handler/llm/routing.rs b/lib/crates/fabro-workflow/src/handler/llm/routing.rs new file mode 100644 index 000000000..1c1dfba5d --- /dev/null +++ b/lib/crates/fabro-workflow/src/handler/llm/routing.rs @@ -0,0 +1,51 @@ +use fabro_graphviz::graph::{self, Node}; +use fabro_types::LlmBackend; + +use super::cli::is_cli_only_model; +use crate::error::Error; + +pub(crate) fn select_run_backend(node: &Node) -> Result { + match node.llm_backend() { + None => { + if node.model().is_some_and(is_cli_only_model) { + Ok(LlmBackend::Cli) + } else { + Ok(LlmBackend::Api) + } + } + Some(Ok(backend)) => Ok(backend), + Some(Err(_)) => Err(unsupported_backend_error( + node.backend().unwrap_or_default(), + )), + } +} + +pub(crate) fn select_one_shot_backend(node: &Node) -> Result { + match node.llm_backend() { + Some(Ok(LlmBackend::Acp)) => Ok(LlmBackend::Acp), + Some(Ok(LlmBackend::Api | LlmBackend::Cli)) | None => Ok(LlmBackend::Api), + Some(Err(_)) => Err(unsupported_backend_error( + node.backend().unwrap_or_default(), + )), + } +} + +pub(crate) fn node_needs_api_backend(node: &Node) -> bool { + if !graph::is_llm_handler_type(node.handler_type()) { + return false; + } + + match node.handler_type() { + Some("prompt" | "one_shot") => { + !matches!(select_one_shot_backend(node), Ok(LlmBackend::Acp)) + } + _ => matches!(select_run_backend(node), Ok(LlmBackend::Api)), + } +} + +fn unsupported_backend_error(raw: &str) -> Error { + Error::Validation(format!( + "unsupported LLM backend \"{raw}\"; expected one of: {}", + LlmBackend::expected_values() + )) +} diff --git a/lib/crates/fabro-workflow/src/handler/prompt.rs b/lib/crates/fabro-workflow/src/handler/prompt.rs index d62654241..c65744741 100644 --- a/lib/crates/fabro-workflow/src/handler/prompt.rs +++ b/lib/crates/fabro-workflow/src/handler/prompt.rs @@ -6,7 +6,8 @@ use fabro_graphviz::graph::{Graph, Node}; use fabro_model::Provider; use super::agent::{ - CodergenBackend, CodergenResult, expand_variables, extract_status_fields, truncate, + CodergenBackend, CodergenResult, OneShotRequest, expand_variables, extract_status_fields, + truncate, }; use super::{EngineServices, Handler}; use crate::context::{Context, WorkflowContext, keys}; @@ -120,13 +121,15 @@ impl Handler for PromptHandler { let (response_text, stage_usage, backend_files_touched) = if let Some(backend) = &self.backend { let result = backend - .one_shot( + .one_shot(OneShotRequest { node, - &prompt, - system_prompt.as_deref(), - &services.run.emitter, - &stage_scope, - ) + prompt: &prompt, + system_prompt: system_prompt.as_deref(), + emitter: &services.run.emitter, + stage_scope: &stage_scope, + sandbox: &services.run.sandbox, + cancel_token: services.run.cancel_token(), + }) .await; match result { Ok(CodergenResult::Full(outcome)) => return Ok(outcome), @@ -207,10 +210,10 @@ mod tests { use fabro_types::fixtures; use object_store::memory::InMemory; use tempfile::TempDir; - use tokio_util::sync::CancellationToken; use super::*; use crate::event::Emitter; + use crate::handler::agent::CodergenRunRequest; fn make_services() -> EngineServices { EngineServices::test_default() @@ -308,33 +311,17 @@ mod tests { #[tokio::test] async fn prompt_handler_dispatches_to_backend_one_shot() { - use fabro_agent::Sandbox; - struct OneShotBackend; #[async_trait] impl CodergenBackend for OneShotBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { panic!("run() should not be called for prompt handler"); } async fn one_shot( &self, - _node: &Node, - _prompt: &str, - _system_prompt: Option<&str>, - _emitter: &Arc, - _stage_scope: &StageScope, + _request: OneShotRequest<'_>, ) -> Result { Ok(CodergenResult::Text { text: "one-shot response".to_string(), @@ -371,33 +358,17 @@ mod tests { #[tokio::test] async fn prompt_handler_projects_provider_used_from_prompt_events() { - use fabro_agent::Sandbox; - struct ProviderOneShotBackend; #[async_trait] impl CodergenBackend for ProviderOneShotBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { panic!("run() should not be called for prompt handler"); } async fn one_shot( &self, - _node: &Node, - _prompt: &str, - _system_prompt: Option<&str>, - _emitter: &Arc, - _stage_scope: &StageScope, + _request: OneShotRequest<'_>, ) -> Result { Ok(CodergenResult::Text { text: "one-shot response".to_string(), @@ -437,30 +408,14 @@ mod tests { #[async_trait] impl CodergenBackend for OneShotCapturingBackend { - async fn run( - &self, - _node: &Node, - _prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: CancellationToken, - ) -> Result { + async fn run(&self, _request: CodergenRunRequest<'_>) -> Result { panic!("run() should not be called for prompt handler"); } - async fn one_shot( - &self, - _node: &Node, - prompt: &str, - system_prompt: Option<&str>, - _emitter: &Arc, - _stage_scope: &StageScope, - ) -> Result { - *self.captured_prompt.lock().unwrap() = Some(prompt.to_string()); - *self.captured_system_prompt.lock().unwrap() = Some(system_prompt.map(String::from)); + async fn one_shot(&self, request: OneShotRequest<'_>) -> Result { + *self.captured_prompt.lock().unwrap() = Some(request.prompt.to_string()); + *self.captured_system_prompt.lock().unwrap() = + Some(request.system_prompt.map(String::from)); Ok(CodergenResult::Text { text: "classified".to_string(), usage: None, diff --git a/lib/crates/fabro-workflow/src/operations/fork.rs b/lib/crates/fabro-workflow/src/operations/fork.rs index ebe8eaed5..2341ddd5b 100644 --- a/lib/crates/fabro-workflow/src/operations/fork.rs +++ b/lib/crates/fabro-workflow/src/operations/fork.rs @@ -231,6 +231,9 @@ fn replay_event_for_fork_projection(body: &EventBody) -> bool { | EventBody::AgentCliStarted(_) | EventBody::AgentCliCancelled(_) | EventBody::AgentCliTimedOut(_) + | EventBody::AgentAcpStarted(_) + | EventBody::AgentAcpCancelled(_) + | EventBody::AgentAcpTimedOut(_) | EventBody::CommandStarted(_) | EventBody::CommandCompleted(_) | EventBody::ParallelCompleted(_) @@ -316,6 +319,41 @@ mod tests { )); } + #[test] + fn fork_replay_preserves_agent_acp_projection_events() { + assert!(replay_event_for_fork_projection( + &EventBody::AgentAcpStarted(fabro_types::run_event::AgentAcpStartedProps { + visit: 1, + mode: "acp".to_string(), + provider: "openai".to_string(), + model: "fake-acp".to_string(), + command: "python fake_agent.py".to_string(), + }) + )); + assert!(replay_event_for_fork_projection( + &EventBody::AgentAcpCancelled(fabro_types::run_event::AgentAcpCancelledProps { + stdout: "partial".to_string(), + stderr: "cancelled".to_string(), + duration_ms: 7, + }) + )); + assert!(replay_event_for_fork_projection( + &EventBody::AgentAcpTimedOut(fabro_types::run_event::AgentAcpTimedOutProps { + stdout: "partial".to_string(), + stderr: "timeout".to_string(), + duration_ms: 99, + }) + )); + assert!(!replay_event_for_fork_projection( + &EventBody::AgentAcpCompleted(fabro_types::run_event::AgentAcpCompletedProps { + stdout: "done".to_string(), + stderr: String::new(), + stop_reason: "end_turn".to_string(), + duration_ms: 42, + }) + )); + } + #[tokio::test] async fn fork_persists_historical_node_projection_through_target_checkpoint() { let store = test_store(); diff --git a/lib/crates/fabro-workflow/src/operations/start.rs b/lib/crates/fabro-workflow/src/operations/start.rs index a0d65ecf3..0cee89dbc 100644 --- a/lib/crates/fabro-workflow/src/operations/start.rs +++ b/lib/crates/fabro-workflow/src/operations/start.rs @@ -547,6 +547,8 @@ fn runtime_mcp_server(settings: &ResolvedMcpServerSettings) -> McpServerSettings env: env.clone(), }, }, + current_dir: None, + clear_env: false, startup_timeout_secs: settings.startup_timeout_secs, tool_timeout_secs: settings.tool_timeout_secs, } diff --git a/lib/crates/fabro-workflow/src/pipeline/initialize.rs b/lib/crates/fabro-workflow/src/pipeline/initialize.rs index 79ee3c3f7..538d2103d 100644 --- a/lib/crates/fabro-workflow/src/pipeline/initialize.rs +++ b/lib/crates/fabro-workflow/src/pipeline/initialize.rs @@ -28,7 +28,9 @@ use crate::devcontainer_bridge::{devcontainer_to_snapshot_config, run_devcontain use crate::error::Error; use crate::event::{Event, RunNoticeCode, RunNoticeLevel}; use crate::github_token_source::{AppIatMinter, GitHubTokenSource}; -use crate::handler::llm::{AgentApiBackend, AgentCliBackend, BackendRouter}; +use crate::handler::llm::{ + AgentAcpBackend, AgentApiBackend, AgentCliBackend, BackendRouter, routing, +}; use crate::handler::{HandlerRegistry, default_registry}; use crate::run_metadata::{RunMetadataRuntime, build_metadata_writer, metadata_branch_name}; use crate::run_options::{GitCheckpointOptions, RunOptions}; @@ -124,7 +126,13 @@ async fn build_registry( llm_source: Arc, cli_resolver: Option, ) -> Result<(Arc, bool), Error> { - let build_no_backend = || Arc::new(default_registry(Arc::clone(&interviewer), || None)); + let no_backend_interviewer = Arc::clone(&interviewer); + let build_no_backend = move || { + Arc::new(default_registry( + Arc::clone(&no_backend_interviewer), + || None, + )) + }; if spec.dry_run { return Ok((build_no_backend(), true)); @@ -135,6 +143,51 @@ async fn build_registry( .values() .any(|n| graph::is_llm_handler_type(n.handler_type())); + if !graph_needs_llm { + return Ok((build_no_backend(), false)); + } + + let build_llm_registry = || { + let model = spec.model.clone(); + let provider = spec.provider; + let fallback_chain = spec.fallback_chain.clone(); + let mcp_servers = spec.mcp_servers.clone(); + let llm_source_for_api = Arc::clone(&llm_source); + let steering_hub_for_api = Arc::clone(&steering_hub); + let tool_env_provider_for_backend = Arc::clone(&tool_env_provider); + Arc::new(default_registry(interviewer, move || { + let tool_env_provider = Arc::clone(&tool_env_provider_for_backend); + let api = AgentApiBackend::new( + model.clone(), + provider, + fallback_chain.clone(), + Arc::clone(&llm_source_for_api), + Arc::clone(&steering_hub_for_api), + ) + .with_tool_env_provider(tool_env_provider.clone()) + .with_mcp_servers(mcp_servers.clone()); + let cli = cli_resolver + .clone() + .map_or_else( + || AgentCliBackend::new_from_env(model.clone(), provider), + |resolver| AgentCliBackend::new(model.clone(), provider, resolver), + ) + .with_tool_env_provider(tool_env_provider.clone(), github_token_refresh_managed); + let acp = cli_resolver + .clone() + .map_or_else( + || AgentAcpBackend::new_from_env(model.clone(), provider), + |resolver| AgentAcpBackend::new(model.clone(), provider, resolver), + ) + .with_tool_env_provider(tool_env_provider.clone(), github_token_refresh_managed); + Some(Box::new(BackendRouter::new(Box::new(api), cli, acp))) + })) + }; + + if !graph_needs_api_backend(graph) { + return Ok((build_llm_registry(), false)); + } + match llm_source.resolve().await { Ok(result) if result.credentials.is_empty() => { if graph_needs_llm { @@ -156,36 +209,7 @@ async fn build_registry( } Ok((build_no_backend(), false)) } - Ok(_result) => { - let model = spec.model.clone(); - let provider = spec.provider; - let fallback_chain = spec.fallback_chain.clone(); - let mcp_servers = spec.mcp_servers.clone(); - let llm_source_for_api = Arc::clone(&llm_source); - let steering_hub_for_api = Arc::clone(&steering_hub); - let tool_env_provider_for_backend = Arc::clone(&tool_env_provider); - let registry = Arc::new(default_registry(interviewer, move || { - let tool_env_provider = Arc::clone(&tool_env_provider_for_backend); - let api = AgentApiBackend::new( - model.clone(), - provider, - fallback_chain.clone(), - Arc::clone(&llm_source_for_api), - Arc::clone(&steering_hub_for_api), - ) - .with_tool_env_provider(tool_env_provider.clone()) - .with_mcp_servers(mcp_servers.clone()); - let cli = cli_resolver - .clone() - .map_or_else( - || AgentCliBackend::new_from_env(model.clone(), provider), - |resolver| AgentCliBackend::new(model.clone(), provider, resolver), - ) - .with_tool_env_provider(tool_env_provider, github_token_refresh_managed); - Some(Box::new(BackendRouter::new(Box::new(api), cli))) - })); - Ok((registry, false)) - } + Ok(_result) => Ok((build_llm_registry(), false)), Err(e) => { if graph_needs_llm { return Err(Error::Precondition(format!( @@ -197,6 +221,10 @@ async fn build_registry( } } +fn graph_needs_api_backend(graph: &graph::Graph) -> bool { + graph.nodes.values().any(routing::node_needs_api_backend) +} + fn build_llm_source(vault: Option>>) -> Arc { match vault { Some(vault) => Arc::new(VaultCredentialSource::new(vault)), @@ -666,6 +694,7 @@ mod tests { use std::sync::Arc; use std::time::Duration; + use fabro_acp::test_support::fake_acp_agent_script; use fabro_auth::{AuthCredential, AuthDetails}; use fabro_graphviz::graph::{AttrValue, Edge, Graph, Node}; use fabro_interview::AutoApproveInterviewer; @@ -674,9 +703,11 @@ mod tests { use fabro_types::{EventBody, RunEvent, RunId, WorkflowSettings, fixtures}; use fabro_vault::{SecretType, Vault}; use object_store::memory::InMemory; + use tokio::fs::{create_dir_all, write}; use tokio::sync::RwLock as AsyncRwLock; use super::*; + use crate::context::{Context, keys}; use crate::event::StoreProgressLogger; use crate::pipeline::types::InitOptions; use crate::records::RunSpec; @@ -997,6 +1028,170 @@ mod tests { assert!(!effective_dry_run); } + #[tokio::test] + async fn initialize_executes_acp_backend_node_from_registry() { + let temp = tempfile::tempdir().unwrap(); + let run_dir = temp.path().join("run"); + create_dir_all(&run_dir).await.unwrap(); + let script_path = temp.path().join("fake_acp_agent.py"); + write(&script_path, fake_acp_agent_script()).await.unwrap(); + + let source = format!( + r#"digraph test {{ + start [shape=Mdiamond]; + writer [type="agent", backend="acp", provider="openai", model="fake-acp", prompt="write hello", acp_command="python3 {}"]; + exit [shape=Msquare]; + start -> writer; + writer -> exit; +}}"#, + script_path.display() + ); + let mut graph = Graph::new("test"); + let mut start = Node::new("start"); + start.attrs.insert( + "shape".to_string(), + AttrValue::String("Mdiamond".to_string()), + ); + let mut writer = Node::new("writer"); + writer + .attrs + .insert("type".to_string(), AttrValue::String("agent".to_string())); + writer + .attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + writer.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + writer.attrs.insert( + "model".to_string(), + AttrValue::String("fake-acp".to_string()), + ); + writer.attrs.insert( + "prompt".to_string(), + AttrValue::String("write hello".to_string()), + ); + writer.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + fabro_sandbox::shell_quote(&script_path.to_string_lossy()) + )), + ); + let mut exit = Node::new("exit"); + exit.attrs.insert( + "shape".to_string(), + AttrValue::String("Msquare".to_string()), + ); + graph.nodes.insert("start".to_string(), start); + graph.nodes.insert("writer".to_string(), writer); + graph.nodes.insert("exit".to_string(), exit); + graph.edges.push(Edge::new("start", "writer")); + graph.edges.push(Edge::new("writer", "exit")); + + let mut vault = Vault::load(temp.path().join("secrets.json")).unwrap(); + vault + .set( + "openai", + &serde_json::to_string(&AuthCredential { + provider: fabro_llm::Provider::OpenAi, + details: AuthDetails::ApiKey { + key: "openai-key".to_string(), + }, + }) + .unwrap(), + SecretType::Credential, + None, + ) + .unwrap(); + let vault = Arc::new(AsyncRwLock::new(vault)); + + let emitter = Arc::new(crate::event::Emitter::new(test_run_id())); + let seen = Arc::new(std::sync::Mutex::new(Vec::new())); + emitter.on_event({ + let seen = Arc::clone(&seen); + move |event| seen.lock().unwrap().push(event.event_name().to_string()) + }); + let store = memory_store(); + let run_store = store.create_run(&test_run_id()).await.unwrap(); + let initialized = initialize(test_persisted(graph, source, &run_dir), InitOptions { + run_id: test_run_id(), + run_store: run_store.into(), + dry_run: false, + emitter: emitter.clone(), + sandbox: SandboxSpec::Local { + working_directory: temp.path().to_path_buf(), + }, + llm: LlmSpec { + model: "fake-acp".to_string(), + provider: fabro_llm::Provider::OpenAi, + fallback_chain: Vec::new(), + mcp_servers: Vec::new(), + dry_run: false, + }, + interviewer: Arc::new(AutoApproveInterviewer::engine()), + steering_hub: Arc::new(crate::steering_hub::SteeringHub::new(emitter)), + lifecycle: crate::run_options::LifecycleOptions { + setup_commands: Vec::new(), + setup_command_timeout_ms: 1_000, + devcontainer_phases: Vec::new(), + }, + run_options: test_settings(&run_dir), + workflow_path: None, + workflow_bundle: None, + hooks: fabro_hooks::HookSettings { hooks: vec![] }, + sandbox_env: SandboxEnvSpec { + devcontainer_env: HashMap::new(), + toml_env: HashMap::new(), + github_permissions: None, + origin_url: None, + }, + vault: Some(vault), + devcontainer: None, + git: None, + run_control: None, + registry_override: None, + artifact_sink: None, + checkpoint: None, + seed_context: None, + }) + .await + .unwrap(); + + let node = initialized.graph.nodes.get("writer").unwrap().clone(); + let handler = initialized.engine.registry.resolve(&node); + let context = Context::new(); + context.set( + keys::INTERNAL_RUN_ID, + serde_json::json!(test_run_id().to_string()), + ); + let outcome = handler + .execute( + &node, + &context, + &initialized.graph, + &initialized.run_options.run_dir, + &initialized.engine, + ) + .await + .unwrap(); + + assert_eq!( + outcome.context_updates.get(&keys::response_key("writer")), + Some(&serde_json::json!("hello from acp")) + ); + assert!( + seen.lock() + .unwrap() + .contains(&"agent.acp.started".to_string()) + ); + assert!( + seen.lock() + .unwrap() + .contains(&"agent.acp.completed".to_string()) + ); + } + #[tokio::test] async fn initialize_runs_setup_commands() { let temp = tempfile::tempdir().unwrap(); diff --git a/lib/crates/fabro-workflow/src/transforms/import.rs b/lib/crates/fabro-workflow/src/transforms/import.rs index 464692846..cce49e9c6 100644 --- a/lib/crates/fabro-workflow/src/transforms/import.rs +++ b/lib/crates/fabro-workflow/src/transforms/import.rs @@ -534,6 +534,7 @@ impl ImportTransform { | "reasoning_effort" | "speed" | "backend" + | "acp_command" | "fidelity" | "max_retries" | "thread_id" @@ -732,7 +733,7 @@ mod tests { let graph = apply_import( r#"digraph Deploy { start [shape=Mdiamond] - validate [import="./validate.fabro", model="haiku", class="fast, shared"] + validate [import="./validate.fabro", model="haiku", backend="acp", acp_command="python fake_agent.py", class="fast, shared"] exit [shape=Msquare] start -> validate -> exit }"#, @@ -772,6 +773,20 @@ mod tests { .iter() .any(|class_name| class_name == "validate") ); + assert_eq!( + graph.nodes["validate.test"] + .attrs + .get("backend") + .and_then(AttrValue::as_str), + Some("acp") + ); + assert_eq!( + graph.nodes["validate.test"] + .attrs + .get("acp_command") + .and_then(AttrValue::as_str), + Some("python fake_agent.py") + ); } #[test] diff --git a/lib/crates/fabro-workflow/tests/it/cp_integration.rs b/lib/crates/fabro-workflow/tests/it/cp_integration.rs index 3aba03426..897b775c3 100644 --- a/lib/crates/fabro-workflow/tests/it/cp_integration.rs +++ b/lib/crates/fabro-workflow/tests/it/cp_integration.rs @@ -17,6 +17,9 @@ use fabro_sandbox::reconnect::reconnect; use fabro_types::{RunSandbox, RunSandboxRuntime, SandboxProvider}; +const DOCKER_MANAGED_LABEL: &str = "sh.fabro.managed"; +const DOCKER_CP_IMAGE: &str = "buildpack-deps:noble"; + // --------------------------------------------------------------------------- // Local sandbox // --------------------------------------------------------------------------- @@ -143,14 +146,84 @@ fn docker_record(container_id: &str) -> RunSandbox { } } +struct DockerCpContainer { + id: String, + cleanup: bool, +} + +impl Drop for DockerCpContainer { + fn drop(&mut self) { + if self.cleanup { + let _ = std::process::Command::new("docker") + .args(["rm", "-f", &self.id]) + .output(); + } + } +} + +fn docker_cp_container() -> DockerCpContainer { + if let Ok(id) = std::env::var("FABRO_DOCKER_CP_CONTAINER") { + return DockerCpContainer { id, cleanup: false }; + } + + ensure_docker_image(DOCKER_CP_IMAGE); + let output = std::process::Command::new("docker") + .args([ + "run", + "-d", + "--label", + &format!("{DOCKER_MANAGED_LABEL}=true"), + "--workdir", + "/workspace", + DOCKER_CP_IMAGE, + "sh", + "-c", + "mkdir -p /workspace && sleep 300", + ]) + .output() + .expect("docker run should execute"); + assert!( + output.status.success(), + "docker run failed\nstdout:\n{}\nstderr:\n{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + let id = String::from_utf8(output.stdout) + .expect("docker run stdout should be UTF-8") + .trim() + .to_string(); + assert!(!id.is_empty(), "docker run should return a container id"); + DockerCpContainer { id, cleanup: true } +} + +fn ensure_docker_image(image: &str) { + let inspect = std::process::Command::new("docker") + .args(["image", "inspect", image]) + .output() + .expect("docker image inspect should execute"); + if inspect.status.success() { + return; + } + + let pull = std::process::Command::new("docker") + .args(["pull", image]) + .output() + .expect("docker pull should execute"); + assert!( + pull.status.success(), + "docker pull {image} failed\nstdout:\n{}\nstderr:\n{}", + String::from_utf8_lossy(&pull.stdout), + String::from_utf8_lossy(&pull.stderr) + ); +} + #[tokio::test] #[ignore] // requires Docker daemon async fn docker_cp_upload_download_round_trip() { - let container_id = std::env::var("FABRO_DOCKER_CP_CONTAINER") - .expect("set FABRO_DOCKER_CP_CONTAINER to an initialized Fabro-managed container ID"); + let container = docker_cp_container(); let scratch = tempfile::tempdir().unwrap(); - let record = docker_record(&container_id); + let record = docker_record(&container.id); let sandbox = reconnect(&record, None).await.expect("reconnect docker"); // Upload a text file @@ -176,11 +249,10 @@ async fn docker_cp_upload_download_round_trip() { #[tokio::test] #[ignore] // requires Docker daemon async fn docker_cp_binary_round_trip() { - let container_id = std::env::var("FABRO_DOCKER_CP_CONTAINER") - .expect("set FABRO_DOCKER_CP_CONTAINER to an initialized Fabro-managed container ID"); + let container = docker_cp_container(); let scratch = tempfile::tempdir().unwrap(); - let record = docker_record(&container_id); + let record = docker_record(&container.id); let sandbox = reconnect(&record, None).await.expect("reconnect docker"); let binary: Vec = (0..=255).collect(); @@ -204,11 +276,10 @@ async fn docker_cp_binary_round_trip() { #[tokio::test] #[ignore] // requires Docker daemon async fn docker_cp_creates_parent_dirs() { - let container_id = std::env::var("FABRO_DOCKER_CP_CONTAINER") - .expect("set FABRO_DOCKER_CP_CONTAINER to an initialized Fabro-managed container ID"); + let container = docker_cp_container(); let scratch = tempfile::tempdir().unwrap(); - let record = docker_record(&container_id); + let record = docker_record(&container.id); let sandbox = reconnect(&record, None).await.expect("reconnect docker"); let content = b"nested docker file\n"; diff --git a/lib/crates/fabro-workflow/tests/it/daytona_integration.rs b/lib/crates/fabro-workflow/tests/it/daytona_integration.rs index 50e5bd39c..c74725414 100644 --- a/lib/crates/fabro-workflow/tests/it/daytona_integration.rs +++ b/lib/crates/fabro-workflow/tests/it/daytona_integration.rs @@ -995,7 +995,7 @@ async fn daytona_parallel_git_branching_e2e() { // CLI Backend on Daytona — real CLI tools via exec_command // --------------------------------------------------------------------------- -use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult}; +use fabro_workflow::handler::agent::{CodergenBackend, CodergenResult, CodergenRunRequest}; use fabro_workflow::handler::llm::AgentCliBackend; /// Helper: run a real CLI backend test on Daytona. @@ -1072,16 +1072,16 @@ async fn run_daytona_cli_test(provider: Provider, model: &str, install_command: let emitter = Arc::new(Emitter::default()); let result = backend - .run( - &node, - "What is 2+2? Reply with just the number.", - &context, - None, - &emitter, - &env, - None, - CancellationToken::new(), - ) + .run(CodergenRunRequest { + node: &node, + prompt: "What is 2+2? Reply with just the number.", + context: &context, + thread_id: None, + emitter: &emitter, + sandbox: &env, + tool_hooks: None, + cancel_token: CancellationToken::new(), + }) .await; match result { @@ -2149,6 +2149,8 @@ async fn daytona_playwright_mcp_sandbox_transport() { port: mcp_port, env: std::collections::HashMap::new(), }, + current_dir: None, + clear_env: false, startup_timeout_secs: 30, tool_timeout_secs: 120, }; @@ -2205,6 +2207,8 @@ async fn daytona_playwright_mcp_sandbox_transport() { fabro_mcp::config::McpServerSettings { name: mcp_config.name.clone(), transport: fabro_mcp::config::McpTransport::Http { url, headers }, + current_dir: mcp_config.current_dir.clone(), + clear_env: mcp_config.clear_env, startup_timeout_secs: mcp_config.startup_timeout_secs, tool_timeout_secs: mcp_config.tool_timeout_secs, } diff --git a/lib/crates/fabro-workflow/tests/it/integration.rs b/lib/crates/fabro-workflow/tests/it/integration.rs index ad659b05a..01324d1c1 100644 --- a/lib/crates/fabro-workflow/tests/it/integration.rs +++ b/lib/crates/fabro-workflow/tests/it/integration.rs @@ -23,6 +23,7 @@ use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::Duration; +use fabro_acp::test_support::fake_acp_agent_script; use fabro_config::RunScratch; use fabro_graphviz::graph::{AttrValue, Edge, Graph, Node}; use fabro_graphviz::parser::parse; @@ -37,11 +38,14 @@ use fabro_validate::{Severity, validate, validate_or_raise}; use fabro_workflow::context::Context; use fabro_workflow::error::{Error, FailureSignatureExt}; use fabro_workflow::event::{Emitter, Event}; -use fabro_workflow::handler::agent::{AgentHandler, CodergenBackend, CodergenResult}; +use fabro_workflow::handler::agent::{ + AgentHandler, CodergenBackend, CodergenResult, CodergenRunRequest, +}; use fabro_workflow::handler::command::CommandHandler; use fabro_workflow::handler::conditional::ConditionalHandler; use fabro_workflow::handler::exit::ExitHandler; use fabro_workflow::handler::human::HumanHandler; +use fabro_workflow::handler::llm::AgentAcpBackend; use fabro_workflow::handler::llm::cli::{AgentCliBackend, BackendRouter, parse_cli_response}; use fabro_workflow::handler::manager_loop::SubWorkflowHandler; use fabro_workflow::handler::start::StartHandler; @@ -63,6 +67,25 @@ fn local_env() -> Arc { )) } +fn codergen_run_request<'a>( + node: &'a Node, + prompt: &'a str, + context: &'a Context, + emitter: &'a Arc, + sandbox: &'a Arc, +) -> CodergenRunRequest<'a> { + CodergenRunRequest { + node, + prompt, + context, + thread_id: None, + emitter, + sandbox, + tool_hooks: None, + cancel_token: CancellationToken::new(), + } +} + fn test_run_id(label: &str) -> RunId { let mut hasher = DefaultHasher::new(); label.hash(&mut hasher); @@ -1604,22 +1627,12 @@ struct MockCodergenBackend; #[async_trait::async_trait] impl CodergenBackend for MockCodergenBackend { - async fn run( - &self, - node: &Node, - prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: tokio_util::sync::CancellationToken, - ) -> Result { + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { Ok(CodergenResult::Text { text: format!( "Response for {}: processed prompt '{}'", - node.id, - &prompt[..prompt.len().min(50)] + request.node.id, + &request.prompt[..request.prompt.len().min(50)] ), usage: None, files_touched: Vec::new(), @@ -6281,9 +6294,10 @@ mod real_llm { use fabro_llm::providers::OpenAiAdapter; use fabro_llm::types::{Message, Request}; use fabro_types::WorkflowSettings; - use fabro_workflow::context::Context; use fabro_workflow::error::Error; - use fabro_workflow::handler::agent::{AgentHandler, CodergenBackend, CodergenResult}; + use fabro_workflow::handler::agent::{ + AgentHandler, CodergenBackend, CodergenResult, CodergenRunRequest, OneShotRequest, + }; use tokio_util::sync::CancellationToken; struct LlmCodergenBackend { @@ -6294,29 +6308,12 @@ mod real_llm { #[async_trait] impl CodergenBackend for LlmCodergenBackend { - async fn run( - &self, - _node: &Node, - prompt: &str, - _context: &Context, - _thread_id: Option<&str>, - _emitter: &Arc, - _sandbox: &Arc, - _tool_hooks: Option>, - _cancel_token: tokio_util::sync::CancellationToken, - ) -> Result { - self.complete(prompt).await + async fn run(&self, request: CodergenRunRequest<'_>) -> Result { + self.complete(request.prompt).await } - async fn one_shot( - &self, - _node: &Node, - prompt: &str, - _system_prompt: Option<&str>, - _emitter: &Arc, - _stage_scope: &fabro_workflow::event::StageScope, - ) -> Result { - self.complete(prompt).await + async fn one_shot(&self, request: OneShotRequest<'_>) -> Result { + self.complete(request.prompt).await } } @@ -9656,14 +9653,25 @@ impl fabro_agent::Sandbox for CliTestEnv { ) -> fabro_sandbox::Result { self.commands.lock().unwrap().push(command.to_string()); - // git diff calls: first pair returns empty (before), second pair returns - // configured files - if command.starts_with("git diff") || command.starts_with("git ls-files") { + // Changed-file snapshot calls: first returns empty (before), second + // returns configured files (after). + if command.contains("__FABRO_CHANGED_FILES_DIFF__") + || command.starts_with("git diff") + || command.starts_with("git ls-files") + { let call_num = self .git_diff_call_count .fetch_add(1, std::sync::atomic::Ordering::SeqCst); - // Calls 0,1 = before snapshot (empty), calls 2,3 = after snapshot - let stdout = if call_num >= 2 && command.starts_with("git diff") { + let stdout = if command.contains("__FABRO_CHANGED_FILES_DIFF__") { + if call_num >= 1 { + format!( + "__FABRO_CHANGED_FILES_DIFF__\n{}__FABRO_CHANGED_FILES_UNTRACKED__\n", + self.git_diff_after + ) + } else { + "__FABRO_CHANGED_FILES_DIFF__\n__FABRO_CHANGED_FILES_UNTRACKED__\n".to_string() + } + } else if call_num >= 2 && command.starts_with("git diff") { self.git_diff_after.clone() } else { String::new() @@ -9678,11 +9686,10 @@ impl fabro_agent::Sandbox for CliTestEnv { }); } - // CLI version check during ensure_cli — return success so install path - // is skipped. - if command.contains("--version") { + // CLI availability check. + if command.contains("command -v ") { return Ok(fabro_agent::ExecResult { - stdout: "1.0.0\n".into(), + stdout: "/usr/local/bin/agent-cli\n".into(), stderr: String::new(), exit_code: Some(0), @@ -9789,24 +9796,20 @@ async fn cli_backend_run_writes_prompt_and_calls_exec() { let claude_output = r#"{"type":"result","result":"I fixed the bug.","usage":{"input_tokens":500,"output_tokens":200}}"#; let test_env = Arc::new(CliTestEnv::new(claude_output)); let env: Arc = test_env.clone(); - let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); let node = Node::new("fix_code"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( + .run(codergen_run_request( &node, "Fix the authentication bug", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await .expect("CLI backend should succeed"); @@ -9863,24 +9866,20 @@ async fn cli_backend_run_detects_changed_files() { let claude_output = r#"{"type":"result","result":"Created new file.","usage":{"input_tokens":100,"output_tokens":50}}"#; let env: Arc = Arc::new(CliTestEnv::new(claude_output).with_git_diff_after("src/main.rs\nsrc/lib.rs\n")); - let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); let node = Node::new("implement"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( + .run(codergen_run_request( &node, "Add a new feature", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await .expect("CLI backend should succeed"); @@ -9897,24 +9896,20 @@ async fn cli_backend_run_with_codex_provider() { let codex_output = "{\"type\":\"item.completed\",\"item\":{\"id\":\"item_0\",\"type\":\"agent_message\",\"text\":\"Implemented the feature.\"}}\n{\"type\":\"turn.completed\",\"usage\":{\"input_tokens\":300,\"output_tokens\":150}}"; let test_env = Arc::new(CliTestEnv::new(codex_output)); let env: Arc = test_env.clone(); - let backend = AgentCliBackend::new_from_env("gpt-5.3-codex".into(), Provider::OpenAi) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("gpt-5.3-codex".into(), Provider::OpenAi); let node = Node::new("implement"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( + .run(codergen_run_request( &node, "Build the API", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await .expect("CLI backend should succeed"); @@ -9991,10 +9986,10 @@ async fn cli_backend_run_fails_on_nonzero_exit() { duration_ms: 0, }); } - // CLI version check during ensure_cli — pretend already installed. - if command.contains("--version") { + // CLI availability check. + if command.contains("command -v ") { return Ok(fabro_agent::ExecResult { - stdout: "1.0.0\n".into(), + stdout: "/usr/local/bin/agent-cli\n".into(), stderr: String::new(), exit_code: Some(0), @@ -10064,8 +10059,7 @@ async fn cli_backend_run_fails_on_nonzero_exit() { } let failing_env: Arc = Arc::new(FailingCliEnv); - let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); let node = Node::new("step"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); @@ -10073,16 +10067,13 @@ async fn cli_backend_run_fails_on_nonzero_exit() { let _ = env; // unused, just for the above struct let result = backend - .run( + .run(codergen_run_request( &node, "do something", &context, - None, &emitter, &failing_env, - None, - CancellationToken::new(), - ) + )) .await; let err = match result { @@ -10103,24 +10094,20 @@ async fn cli_backend_run_fails_on_nonzero_exit() { #[tokio::test] async fn cli_backend_run_fails_on_unparseable_output() { let env: Arc = Arc::new(CliTestEnv::new("this is not json at all")); - let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); let node = Node::new("step"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( + .run(codergen_run_request( &node, "do something", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await; let err = match result { @@ -10140,8 +10127,7 @@ async fn cli_backend_run_uses_node_model_override() { r#"{"type":"result","result":"ok","usage":{"input_tokens":10,"output_tokens":5}}"#; let test_env = Arc::new(CliTestEnv::new(claude_output)); let env: Arc = test_env.clone(); - let backend = AgentCliBackend::new_from_env("default-model".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("default-model".into(), Provider::Anthropic); let mut node = Node::new("step"); node.attrs.insert( @@ -10153,16 +10139,9 @@ async fn cli_backend_run_uses_node_model_override() { let emitter = Arc::new(Emitter::default()); backend - .run( - &node, - "test", - &context, - None, - &emitter, - &env, - None, - CancellationToken::new(), - ) + .run(codergen_run_request( + &node, "test", &context, &emitter, &env, + )) .await .expect("should succeed"); @@ -10186,8 +10165,7 @@ async fn cli_backend_run_uses_node_provider_override() { let codex_output = "{\"type\":\"item.completed\",\"item\":{\"id\":\"item_0\",\"type\":\"agent_message\",\"text\":\"ok\"}}\n{\"type\":\"turn.completed\",\"usage\":{\"input_tokens\":10,\"output_tokens\":5}}"; let test_env = Arc::new(CliTestEnv::new(codex_output)); let env: Arc = test_env.clone(); - let backend = AgentCliBackend::new_from_env("default-model".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("default-model".into(), Provider::Anthropic); let mut node = Node::new("step"); node.attrs.insert( @@ -10203,16 +10181,9 @@ async fn cli_backend_run_uses_node_provider_override() { let emitter = Arc::new(Emitter::default()); backend - .run( - &node, - "test", - &context, - None, - &emitter, - &env, - None, - CancellationToken::new(), - ) + .run(codergen_run_request( + &node, "test", &context, &emitter, &env, + )) .await .expect("should succeed"); @@ -10229,24 +10200,16 @@ async fn cli_backend_run_returns_text_and_usage() { let claude_output = r#"{"type":"result","result":"done","usage":{"input_tokens":10,"output_tokens":5}}"#; let env: Arc = Arc::new(CliTestEnv::new(claude_output)); - let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); + let backend = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); let node = Node::new("step"); let context = Context::new(); let emitter = Arc::new(Emitter::default()); let result = backend - .run( - &node, - "test", - &context, - None, - &emitter, - &env, - None, - CancellationToken::new(), - ) + .run(codergen_run_request( + &node, "test", &context, &emitter, &env, + )) .await .expect("should succeed"); @@ -10264,15 +10227,18 @@ async fn cli_backend_run_returns_text_and_usage() { // -- BackendRouter e2e: delegates to correct backend -- +fn test_acp_backend() -> AgentAcpBackend { + AgentAcpBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) +} + #[tokio::test] async fn backend_router_delegates_to_cli_for_cli_node() { let claude_output = r#"{"type":"result","result":"CLI response","usage":{"input_tokens":10,"output_tokens":5}}"#; let env: Arc = Arc::new(CliTestEnv::new(claude_output)); let api_backend = Box::new(MockCodergenBackend); // would return "Response for ..." - let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - let router = BackendRouter::new(api_backend, cli); + let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + let router = BackendRouter::new(api_backend, cli, test_acp_backend()); let mut node = Node::new("cli_step"); node.attrs @@ -10286,16 +10252,13 @@ async fn backend_router_delegates_to_cli_for_cli_node() { let emitter = Arc::new(Emitter::default()); let result = router - .run( + .run(codergen_run_request( &node, "Fix the bug", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await .expect("router should succeed"); @@ -10315,9 +10278,8 @@ async fn backend_router_delegates_to_api_for_normal_node() { let env = local_env(); let api_backend = Box::new(MockCodergenBackend); - let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - let router = BackendRouter::new(api_backend, cli); + let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + let router = BackendRouter::new(api_backend, cli, test_acp_backend()); let mut node = Node::new("api_step"); node.attrs.insert( @@ -10329,16 +10291,13 @@ async fn backend_router_delegates_to_api_for_normal_node() { let emitter = Arc::new(Emitter::default()); let result = router - .run( + .run(codergen_run_request( &node, "Plan the work", &context, - None, &emitter, &env, - None, - CancellationToken::new(), - ) + )) .await .expect("router should succeed"); @@ -10359,9 +10318,8 @@ async fn backend_router_delegates_to_cli_for_backend_attr() { let env: Arc = Arc::new(CliTestEnv::new(codex_output)); let api_backend = Box::new(MockCodergenBackend); - let cli = AgentCliBackend::new_from_env("gpt-5.3-codex".into(), Provider::OpenAi) - .with_poll_interval(Duration::from_millis(10)); - let router = BackendRouter::new(api_backend, cli); + let cli = AgentCliBackend::new_from_env("gpt-5.3-codex".into(), Provider::OpenAi); + let router = BackendRouter::new(api_backend, cli, test_acp_backend()); let mut node = Node::new("codex_step"); node.attrs @@ -10375,16 +10333,9 @@ async fn backend_router_delegates_to_cli_for_backend_attr() { let emitter = Arc::new(Emitter::default()); let result = router - .run( - &node, - "Build it", - &context, - None, - &emitter, - &env, - None, - CancellationToken::new(), - ) + .run(codergen_run_request( + &node, "Build it", &context, &emitter, &env, + )) .await .expect("router should succeed"); @@ -10399,6 +10350,64 @@ async fn backend_router_delegates_to_cli_for_backend_attr() { } } +#[tokio::test] +async fn backend_router_delegates_to_acp_for_acp_node() { + let tempdir = tempfile::tempdir().unwrap(); + let script_path = tempdir.path().join("fake_acp_agent.py"); + tokio::fs::write(&script_path, fake_acp_agent_script()) + .await + .unwrap(); + let env: Arc = + Arc::new(fabro_agent::LocalSandbox::new(tempdir.path().to_path_buf())); + + let api_backend = Box::new(MockCodergenBackend); + let cli = AgentCliBackend::new_from_env("gpt-5.3-codex".into(), Provider::OpenAi); + let router = BackendRouter::new( + api_backend, + cli, + AgentAcpBackend::new_from_env("fake-acp".into(), Provider::OpenAi), + ); + + let mut node = Node::new("acp_step"); + node.attrs + .insert("backend".to_string(), AttrValue::String("acp".to_string())); + node.attrs.insert( + "provider".to_string(), + AttrValue::String("openai".to_string()), + ); + node.attrs.insert( + "model".to_string(), + AttrValue::String("fake-acp".to_string()), + ); + node.attrs.insert( + "acp_command".to_string(), + AttrValue::String(format!( + "python3 {}", + fabro_agent::shell_quote(&script_path.to_string_lossy()) + )), + ); + + let context = Context::new(); + let emitter = Arc::new(Emitter::default()); + + let result = router + .run(codergen_run_request( + &node, "Build it", &context, &emitter, &env, + )) + .await + .expect("router should succeed"); + + match result { + CodergenResult::Text { text, .. } => { + assert_eq!( + text, "hello from acp", + "should route to ACP backend for backend=acp" + ); + } + CodergenResult::Full(_) => panic!("expected Text result"), + } +} + // -- Full pipeline e2e with BackendRouter -- #[tokio::test] @@ -10453,9 +10462,8 @@ async fn full_pipeline_with_cli_backend_node() { // Build engine with BackendRouter let api = MockCodergenBackend; - let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - let router = BackendRouter::new(Box::new(api), cli); + let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + let router = BackendRouter::new(Box::new(api), cli, test_acp_backend()); let codergen_handler = AgentHandler::new(Some(Box::new(router))); let mut registry = HandlerRegistry::new(Box::new(codergen_handler)); @@ -10466,9 +10474,8 @@ async fn full_pipeline_with_cli_backend_node() { Box::new(AgentHandler::new(Some(Box::new({ // Second BackendRouter for the "agent" handler let api2 = MockCodergenBackend; - let cli2 = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - BackendRouter::new(Box::new(api2), cli2) + let cli2 = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + BackendRouter::new(Box::new(api2), cli2, test_acp_backend()) })))), ); @@ -10575,17 +10582,15 @@ async fn stylesheet_backend_property_routes_to_cli() { // Run the pipeline let api = MockCodergenBackend; - let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - let router = BackendRouter::new(Box::new(api), cli); + let cli = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + let router = BackendRouter::new(Box::new(api), cli, test_acp_backend()); let mut registry = HandlerRegistry::new(Box::new(AgentHandler::new(Some(Box::new(router))))); registry.register("start", Box::new(StartHandler)); registry.register("exit", Box::new(ExitHandler)); let api2 = MockCodergenBackend; - let cli2 = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic) - .with_poll_interval(Duration::from_millis(10)); - let router2 = BackendRouter::new(Box::new(api2), cli2); + let cli2 = AgentCliBackend::new_from_env("claude-opus-4-6".into(), Provider::Anthropic); + let router2 = BackendRouter::new(Box::new(api2), cli2, test_acp_backend()); registry.register( "agent", Box::new(AgentHandler::new(Some(Box::new(router2)))),