From 24f0fa7ca0b5a7098a085be4a293227267ff93b0 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Thu, 5 Mar 2026 18:31:19 -0500 Subject: [PATCH] Add Daytona network access control and fill in execution docs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Add DaytonaNetwork enum (block/allow_all/allow_list) with custom serde Deserialize for TOML string-or-table syntax - Wire network config through run_config defaults merging and base_params - Document network access in sandboxing, environments, and run-configuration - Fill in execution docs: checkpoints, environments, failures, interviews, run configuration, observability, retros - Rename compounding.mdx → retros.mdx, insights.mdx → observability.mdx - Use DaytonaConfig::default() in tests to reduce boilerplate Co-Authored-By: Claude Opus 4.6 --- crates/arc-workflows/src/cli/run_config.rs | 101 +++++- crates/arc-workflows/src/daytona_sandbox.rs | 178 +++++++++++ .../tests/daytona_integration.rs | 2 +- docs/administration/sandboxing.mdx | 11 + docs/agents/artifacts.mdx | 5 + docs/core-concepts/how-arc-works.mdx | 2 +- docs/docs.json | 7 +- docs/execution/checkpoints.mdx | 161 +++++++++- docs/execution/compounding.mdx | 5 - docs/execution/environments.mdx | 217 ++++++++++++- docs/execution/failures.mdx | 262 +++++++++++++++- docs/execution/interviews.mdx | 153 +++++++++- .../{insights.mdx => observability.mdx} | 2 +- docs/execution/retros.mdx | 186 +++++++++++ docs/execution/run-configuration.mdx | 289 +++++++++++++++++- docs/workflows/stages-and-nodes.mdx | 4 +- 16 files changed, 1554 insertions(+), 31 deletions(-) create mode 100644 docs/agents/artifacts.mdx delete mode 100644 docs/execution/compounding.mdx rename docs/execution/{insights.mdx => observability.mdx} (80%) create mode 100644 docs/execution/retros.mdx diff --git a/crates/arc-workflows/src/cli/run_config.rs b/crates/arc-workflows/src/cli/run_config.rs index 6017c3515..41658359f 100644 --- a/crates/arc-workflows/src/cli/run_config.rs +++ b/crates/arc-workflows/src/cli/run_config.rs @@ -115,6 +115,9 @@ impl WorkflowRunConfig { } task_d.labels = Some(merged); } + if task_d.network.is_none() { + task_d.network = default_d.network.clone(); + } } (None, Some(_)) => task.daytona = default.daytona.clone(), _ => {} @@ -778,8 +781,7 @@ provider = "daytona" preserve: None, daytona: Some(DaytonaConfig { auto_stop_interval: Some(30), - labels: None, - snapshot: None, + ..DaytonaConfig::default() }), }), ..RunDefaults::default() @@ -814,7 +816,7 @@ auto_stop_interval = 60 daytona: Some(DaytonaConfig { auto_stop_interval: Some(30), labels: Some(HashMap::from([("env".into(), "prod".into())])), - snapshot: None, + ..DaytonaConfig::default() }), }), ..RunDefaults::default() @@ -844,12 +846,11 @@ env = "from_task" provider: None, preserve: None, daytona: Some(DaytonaConfig { - auto_stop_interval: None, labels: Some(HashMap::from([ ("env".into(), "from_default".into()), ("team".into(), "platform".into()), ])), - snapshot: None, + ..DaytonaConfig::default() }), }), ..RunDefaults::default() @@ -880,8 +881,6 @@ cpu = 2 provider: None, preserve: None, daytona: Some(DaytonaConfig { - auto_stop_interval: None, - labels: None, snapshot: Some(DaytonaSnapshotConfig { name: "default-snap".into(), cpu: Some(8), @@ -889,6 +888,7 @@ cpu = 2 disk: Some(100), dockerfile: Some("FROM ubuntu".into()), }), + ..DaytonaConfig::default() }), }), ..RunDefaults::default() @@ -918,8 +918,6 @@ auto_stop_interval = 60 provider: None, preserve: None, daytona: Some(DaytonaConfig { - auto_stop_interval: None, - labels: None, snapshot: Some(DaytonaSnapshotConfig { name: "default-snap".into(), cpu: Some(4), @@ -927,6 +925,7 @@ auto_stop_interval = 60 disk: None, dockerfile: None, }), + ..DaytonaConfig::default() }), }), ..RunDefaults::default() @@ -1120,4 +1119,88 @@ graph = "test.dot" let cfg: WorkflowRunConfig = toml::from_str(toml).unwrap(); assert!(cfg.hooks.is_empty()); } + + #[test] + fn parse_toml_with_daytona_network_block() { + let toml = r#" +version = 1 +goal = "test" +graph = "w.dot" + +[sandbox.daytona] +network = "block" +"#; + let config = parse_run_config(toml).unwrap(); + let daytona = config.sandbox.unwrap().daytona.unwrap(); + assert_eq!( + daytona.network, + Some(crate::daytona_sandbox::DaytonaNetwork::Block) + ); + } + + #[test] + fn apply_defaults_network_task_wins() { + let mut cfg = parse_run_config( + r#" +version = 1 +goal = "test" +graph = "w.dot" + +[sandbox.daytona] +network = "block" +"#, + ) + .unwrap(); + let defaults = RunDefaults { + sandbox: Some(SandboxConfig { + provider: None, + preserve: None, + daytona: Some(DaytonaConfig { + network: Some(crate::daytona_sandbox::DaytonaNetwork::AllowAll), + ..DaytonaConfig::default() + }), + }), + ..RunDefaults::default() + }; + cfg.apply_defaults(&defaults); + assert_eq!( + cfg.sandbox.unwrap().daytona.unwrap().network, + Some(crate::daytona_sandbox::DaytonaNetwork::Block) + ); + } + + #[test] + fn apply_defaults_network_inherited() { + let mut cfg = parse_run_config( + r#" +version = 1 +goal = "test" +graph = "w.dot" + +[sandbox.daytona] +auto_stop_interval = 60 +"#, + ) + .unwrap(); + let defaults = RunDefaults { + sandbox: Some(SandboxConfig { + provider: None, + preserve: None, + daytona: Some(DaytonaConfig { + network: Some(crate::daytona_sandbox::DaytonaNetwork::AllowList(vec![ + "10.0.0.0/8".into(), + ])), + ..DaytonaConfig::default() + }), + }), + ..RunDefaults::default() + }; + cfg.apply_defaults(&defaults); + assert_eq!( + cfg.sandbox.unwrap().daytona.unwrap().network, + Some(crate::daytona_sandbox::DaytonaNetwork::AllowList(vec![ + "10.0.0.0/8".into(), + ])) + ); + } } diff --git a/crates/arc-workflows/src/daytona_sandbox.rs b/crates/arc-workflows/src/daytona_sandbox.rs index 2a2073b71..58117fa4a 100644 --- a/crates/arc-workflows/src/daytona_sandbox.rs +++ b/crates/arc-workflows/src/daytona_sandbox.rs @@ -8,6 +8,7 @@ use arc_agent::sandbox::{ }; use async_trait::async_trait; use rand::Rng; +use serde::de::{self, MapAccess, Visitor}; use serde::Deserialize; const WORKING_DIRECTORY: &str = "/home/daytona/workspace"; @@ -21,6 +22,84 @@ pub struct DaytonaConfig { pub auto_stop_interval: Option, pub labels: Option>, pub snapshot: Option, + pub network: Option, +} + +/// Network access mode for a Daytona sandbox. +/// +/// TOML syntax: +/// ```toml +/// network = "block" # no egress +/// network = "allow_all" # full access (default) +/// network = { allow_list = ["208.80.154.232/32"] } # CIDR allowlist +/// ``` +#[derive(Clone, Debug, PartialEq)] +pub enum DaytonaNetwork { + Block, + AllowAll, + AllowList(Vec), +} + +impl<'de> Deserialize<'de> for DaytonaNetwork { + fn deserialize(deserializer: D) -> Result + where + D: serde::Deserializer<'de>, + { + struct DaytonaNetworkVisitor; + + impl<'de> Visitor<'de> for DaytonaNetworkVisitor { + type Value = DaytonaNetwork; + + fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + formatter, + r#""block", "allow_all", or {{ allow_list = [...] }}"# + ) + } + + fn visit_str(self, value: &str) -> Result { + match value { + "block" => Ok(DaytonaNetwork::Block), + "allow_all" => Ok(DaytonaNetwork::AllowAll), + other => Err(de::Error::custom(format!( + "unknown network mode \"{other}\": expected \"block\" or \"allow_all\"" + ))), + } + } + + fn visit_map>(self, mut map: M) -> Result { + let Some(key) = map.next_key::()? else { + return Err(de::Error::custom( + "empty table: expected { allow_list = [...] }", + )); + }; + + if key != "allow_list" { + return Err(de::Error::custom(format!( + "unknown key \"{key}\": expected \"allow_list\"" + ))); + } + + let cidrs: Vec = map.next_value()?; + + if cidrs.is_empty() { + return Err(de::Error::custom( + "allow_list must not be empty", + )); + } + + if let Some(extra) = map.next_key::()? { + return Err(de::Error::custom(format!( + "unexpected key \"{extra}\": allow_list table must have exactly one key" + ))); + } + + Ok(DaytonaNetwork::AllowList(cidrs)) + } + } + + deserializer.deserialize_any(DaytonaNetworkVisitor) + } } /// Snapshot configuration: when present, the sandbox is created from a snapshot @@ -99,11 +178,19 @@ impl DaytonaSandbox { chrono::Utc::now().format("%Y%m%d-%H%M%S"), rand::thread_rng().gen_range(0..0x10000u32), ); + let (network_block_all, network_allow_list) = match &self.config.network { + Some(DaytonaNetwork::Block) => (Some(true), None), + Some(DaytonaNetwork::AllowAll) => (Some(false), None), + Some(DaytonaNetwork::AllowList(cidrs)) => (None, Some(cidrs.clone())), + None => (None, None), + }; daytona_sdk::SandboxBaseParams { name: Some(name), auto_stop_interval: self.config.auto_stop_interval, labels: self.config.labels.clone(), ephemeral: Some(true), + network_block_all, + network_allow_list, ..Default::default() } } @@ -932,4 +1019,95 @@ mod tests { let token = get_gh_token().unwrap(); assert!(!token.is_empty()); } + + #[test] + fn network_block_from_string() { + let config: DaytonaConfig = toml::from_str(r#"network = "block""#).unwrap(); + assert_eq!(config.network, Some(DaytonaNetwork::Block)); + } + + #[test] + fn network_allow_all_from_string() { + let config: DaytonaConfig = toml::from_str(r#"network = "allow_all""#).unwrap(); + assert_eq!(config.network, Some(DaytonaNetwork::AllowAll)); + } + + #[test] + fn network_allow_list_from_table() { + let config: DaytonaConfig = + toml::from_str(r#"network = { allow_list = ["10.0.0.0/8", "172.16.0.0/12"] }"#) + .unwrap(); + assert_eq!( + config.network, + Some(DaytonaNetwork::AllowList(vec![ + "10.0.0.0/8".into(), + "172.16.0.0/12".into(), + ])) + ); + } + + #[test] + fn network_typo_string_error() { + let err = toml::from_str::(r#"network = "blck""#).unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains(r#"unknown network mode "blck""#), + "unexpected error: {msg}" + ); + } + + #[test] + fn network_wrong_type_error() { + let err = toml::from_str::("network = 42").unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains("expected") && msg.contains("allow_list"), + "unexpected error: {msg}" + ); + } + + #[test] + fn network_unknown_key_error() { + let err = + toml::from_str::(r#"network = { mode = "block" }"#).unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains(r#"unknown key "mode""#), + "unexpected error: {msg}" + ); + } + + #[test] + fn network_empty_table_error() { + let err = toml::from_str::("network = {}").unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains("empty table"), + "unexpected error: {msg}" + ); + } + + #[test] + fn network_empty_allow_list_error() { + let err = + toml::from_str::("network = { allow_list = [] }").unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains("allow_list must not be empty"), + "unexpected error: {msg}" + ); + } + + #[test] + fn network_extra_key_error() { + let err = toml::from_str::( + r#"network = { allow_list = ["10.0.0.0/8"], extra = true }"#, + ) + .unwrap_err(); + let msg = err.to_string(); + assert!( + msg.contains(r#"unexpected key "extra""#), + "unexpected error: {msg}" + ); + } } diff --git a/crates/arc-workflows/tests/daytona_integration.rs b/crates/arc-workflows/tests/daytona_integration.rs index de844e8b9..1cf8dfa9a 100644 --- a/crates/arc-workflows/tests/daytona_integration.rs +++ b/crates/arc-workflows/tests/daytona_integration.rs @@ -126,7 +126,6 @@ async fn daytona_snapshot_sandbox() { let config = DaytonaConfig { auto_stop_interval: Some(60), - labels: None, snapshot: Some(DaytonaSnapshotConfig { name: "arc-test-snapshot".to_string(), cpu: Some(2), @@ -136,6 +135,7 @@ async fn daytona_snapshot_sandbox() { "FROM ubuntu:22.04\nRUN apt-get update && apt-get install -y ripgrep".to_string(), ), }), + ..DaytonaConfig::default() }; let env = DaytonaSandbox::new(client, config); diff --git a/docs/administration/sandboxing.mdx b/docs/administration/sandboxing.mdx index aa351b5bb..6b922c09e 100644 --- a/docs/administration/sandboxing.mdx +++ b/docs/administration/sandboxing.mdx @@ -3,3 +3,14 @@ title: "Sandboxing" description: "Sandboxing workflow execution" --- +Sandboxes isolate agent execution from the host machine. When an agent runs a shell command, edits a file, or searches code, it does so inside a sandbox — preventing unintended side effects on the host and providing a reproducible environment for each run. + +Arc supports three sandbox providers: `local` (no isolation), `docker` (container-level), and `daytona` (cloud VM). See [Environments](/execution/environments) for full provider-specific configuration. + +## Network access control + +For cloud sandboxes (Daytona), you can control outbound network access with the `network` field in `[sandbox.daytona]`. Three modes are available: `"allow_all"` (default), `"block"`, and `{ allow_list = ["..."] }` for CIDR-based egress filtering. + +Server defaults in `server.toml` apply when a run config doesn't specify `network`. Individual run configs can override the server default. + +See [Environments — Network access](/execution/environments#network-access) for syntax examples and the full reference. diff --git a/docs/agents/artifacts.mdx b/docs/agents/artifacts.mdx new file mode 100644 index 000000000..5ffe2aeeb --- /dev/null +++ b/docs/agents/artifacts.mdx @@ -0,0 +1,5 @@ +--- +title: "Artifacts" +description: "File artifacts produced by agent execution" +--- + diff --git a/docs/core-concepts/how-arc-works.mdx b/docs/core-concepts/how-arc-works.mdx index 49ec2f759..8c998d400 100644 --- a/docs/core-concepts/how-arc-works.mdx +++ b/docs/core-concepts/how-arc-works.mdx @@ -72,7 +72,7 @@ Every significant action — stage starts, LLM calls, tool invocations, edge sel - **Retrospectives** generated automatically after each run - **DuckDB queries** via `arc insights` for SQL-based analysis across runs -See [Insights](/execution/insights) for more on querying run data. +See [Observability](/execution/observability) for more on querying run data. ## Resuming runs diff --git a/docs/docs.json b/docs/docs.json index aa66397ee..44b6f3ca6 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -53,8 +53,8 @@ "execution/checkpoints", "execution/interviews", "execution/failures", - "execution/insights", - "execution/compounding" + "execution/retros", + "execution/observability" ] }, { @@ -68,7 +68,8 @@ "agents/hooks", "agents/subagents", "agents/permissions", - "agents/outputs" + "agents/outputs", + "agents/artifacts" ] }, { diff --git a/docs/execution/checkpoints.mdx b/docs/execution/checkpoints.mdx index e8a759090..4eb165a7b 100644 --- a/docs/execution/checkpoints.mdx +++ b/docs/execution/checkpoints.mdx @@ -1,7 +1,164 @@ --- title: "Checkpoints" -description: "Saving and resuming workflow state" +description: "How Arc uses Git to checkpoint and resume workflow runs" --- -## Resume +Arc checkpoints every workflow run using Git. After each node completes, Arc commits the file changes and execution state so that interrupted runs can be resumed exactly where they left off. This happens automatically — no configuration required beyond running inside a Git repository. +## Two branches, two purposes + +Each run creates two Git branches that work in tandem: + +| Branch | Ref format | Contains | +|---|---|---| +| **Run branch** | `arc/run/{run_id}` | File changes made by agents and commands — the actual work product | +| **Metadata branch** | `refs/arc/{run_id}` | Checkpoint JSON, the workflow graph, a run manifest, and offloaded artifacts | + +The run branch is a regular Git branch that grows one commit per completed node. The metadata branch is an orphan branch (no shared history with your code) that stores structured data using Git's object database directly — no working tree needed. + +### Run branch commits + +After each node finishes, Arc stages all file changes and creates a commit on the run branch: + +``` +arc(01JKXYZ...): plan (success) + +Arc-Run: 01JKXYZ... +Arc-Completed: 2 +Arc-Checkpoint: a1b2c3d4... +``` + +The commit message follows a structured format: + +| Part | Description | +|---|---| +| Subject line | `arc({run_id}): {node_id} ({status})` | +| `Arc-Run` trailer | The run ID | +| `Arc-Completed` trailer | Number of completed nodes so far | +| `Arc-Checkpoint` trailer | SHA of the corresponding commit on the metadata branch | + +The `Arc-Checkpoint` trailer links each run branch commit to its metadata branch commit, so you can navigate from file changes to the full execution state and back. + +### Metadata branch + +The metadata branch (`refs/arc/{run_id}`) is an orphan branch that stores structured run data using Git's object storage directly (via `git2`). It is initialized at run start with: + +- **`manifest.json`** — Run metadata: run ID, graph name, node/edge counts, base SHA, and branch name +- **`graph.dot`** — The workflow DOT source as it was parsed + +After each node, the metadata branch is updated with: + +- **`checkpoint.json`** — Full execution state (see below) +- **`artifacts/*.json`** — Any offloaded artifact data (large context values over 100KB) + +## What's in a checkpoint + +The `checkpoint.json` captures everything needed to resume a run: + +| Field | Description | +|---|---| +| `timestamp` | When the checkpoint was created | +| `current_node` | The node that just completed | +| `next_node_id` | The next node the engine would execute | +| `completed_nodes` | Ordered list of all completed node IDs | +| `node_retries` | How many retry attempts each node has used | +| `node_outcomes` | Full outcome (status, context updates, usage) for each completed node | +| `context_values` | Snapshot of the entire [run context](/execution/context) | +| `logs` | Internal log entries | +| `git_commit_sha` | SHA of the run branch commit at this checkpoint | +| `loop_failure_signatures` | Failure signature counts for loop detection | +| `restart_failure_signatures` | Failure signature counts across loop-restart edges | + +The checkpoint is also saved to `checkpoint.json` in the logs directory for quick local access. + +## Worktrees + +Arc uses Git worktrees to isolate workflow runs from your working directory. When a run starts in a clean Git repository: + +1. Arc records the current HEAD as the **base SHA** +2. Creates a new branch `arc/run/{run_id}` at that SHA +3. Adds a worktree at `{logs_dir}/worktree` on that branch +4. Changes into the worktree directory for the duration of the run + +This means your original working directory stays untouched while the agent makes changes in the worktree. When the run completes, Arc removes the worktree and restores your original directory. + + +If the working directory has uncommitted changes, Arc skips worktree setup and runs in place, logging a warning. Git checkpointing is disabled in this case. + + +For Daytona sandboxes, the worktree is created inside the remote sandbox instead. The metadata branch is still written to the host repository so that runs can be resumed locally. + +## Resuming a run + +There are two ways to resume an interrupted run: + +### From a checkpoint file + +Resume from a `checkpoint.json` saved in the logs directory: + +```bash +arc run start workflow.dot --resume path/to/logs/checkpoint.json +``` + +Arc loads the checkpoint, restores the context and execution state, and continues from the next node after the checkpoint. + +### From a run branch + +Resume from the Git branches created during a previous run: + +```bash +arc run start --run-branch arc/run/01JKXYZ... +``` + +This reads the checkpoint, manifest, and graph DOT from the metadata branch (`refs/arc/01JKXYZ...`), re-attaches a worktree to the existing run branch, and resumes execution. No workflow file argument is needed — everything is recovered from Git. + + +1. Arc reads `checkpoint.json` from the metadata branch +2. Reads `manifest.json` and `graph.dot` to reconstruct the workflow +3. Creates a fresh worktree attached to the existing run branch +4. Restores the full context, completed node list, retry counts, and failure signatures +5. If the checkpointed node used `full` fidelity, downgrades the first resumed node to `summary:high` (since the original conversation thread no longer exists in memory) +6. Continues execution from `next_node_id` + + +## The checkpoint cycle + +Here's the full sequence that runs after every node completes: + +1. **Save checkpoint to disk** — Write `checkpoint.json` to the logs directory +2. **Write metadata branch** — Serialize the checkpoint and any new artifacts to the metadata branch (shadow commit) +3. **Commit to run branch** — Stage all file changes, commit with structured trailers linking to the shadow commit SHA +4. **Update checkpoint** — Re-save `checkpoint.json` with the `git_commit_sha` field set + +Steps 2-4 are best-effort — if any Git operation fails, the run continues and logs a warning. The disk checkpoint from step 1 is always available as a fallback. + +## Inspecting run history + +Because checkpoints are plain Git commits, you can inspect them with standard Git tools: + +```bash +# View the commit log for a run +git log arc/run/01JKXYZ... --oneline + +# See what an agent changed at a specific node +git show arc/run/01JKXYZ... + +# Diff the full run against the starting point +git diff main..arc/run/01JKXYZ... + +# Read checkpoint data from the metadata branch +git show refs/arc/01JKXYZ...:checkpoint.json | jq .current_node +``` + +## When checkpointing is active + +Git checkpointing activates automatically when: + +- The working directory is a clean Git repository (local and Docker sandboxes) +- The sandbox is Daytona (metadata branch on the host, commits inside the sandbox) + +It is skipped when: + +- The working directory has uncommitted changes +- The working directory is not a Git repository +- The run uses `--dry-run` diff --git a/docs/execution/compounding.mdx b/docs/execution/compounding.mdx deleted file mode 100644 index 1d91c4eac..000000000 --- a/docs/execution/compounding.mdx +++ /dev/null @@ -1,5 +0,0 @@ ---- -title: "Compounding" -description: "Compounding workflows for iterative improvement" ---- - diff --git a/docs/execution/environments.mdx b/docs/execution/environments.mdx index 8fe141354..dc0af9ef7 100644 --- a/docs/execution/environments.mdx +++ b/docs/execution/environments.mdx @@ -1,5 +1,220 @@ --- title: "Environments" -description: "Execution environments for workflow runs" +description: "Sandbox providers for workflow execution" --- +When an agent runs a shell command, edits a file, or searches code, it does so inside a **sandbox**. The sandbox is the execution environment for all tool operations — it controls where commands run, which files are visible, and how much isolation exists between the agent and the host. + +Arc supports three sandbox providers. Each one implements the same interface (file I/O, command execution, grep, glob), so workflows run identically regardless of which provider you choose. The difference is in where and how the tools execute. + +| Provider | Runs on | Isolation | Use case | +|---|---|---|---| +| `local` | Host machine | None | Development, trusted workflows | +| `docker` | Docker container | Container-level | Reproducible environments, untrusted code | +| `daytona` | Cloud VM | Full machine | CI/CD, team-shared runs, SSH debugging | + +## Choosing a provider + +Set the sandbox provider via CLI flag, [run config TOML](/execution/run-configuration), or server defaults: + +```bash +# CLI flag +arc run start workflow.dot --sandbox local +arc run start workflow.dot --sandbox docker +arc run start workflow.dot --sandbox daytona +``` + +```toml +# Run config TOML +[sandbox] +provider = "daytona" +``` + +The precedence order is: CLI flag > run config TOML > server defaults > built-in default (`local`). + +## Local + +The local sandbox runs all tool operations directly on the host machine. It's the default and the simplest option — no setup required beyond the Arc binary itself. + +### How it works + +- **Working directory** — Set to the current directory (or `directory` from the run config). Arc creates it if it doesn't exist. +- **Commands** — Executed via `/bin/bash -c` in the working directory. +- **File operations** — Read and write directly to the host filesystem. Relative paths resolve against the working directory. +- **Cleanup** — No-op. Local sandbox doesn't create or destroy anything on cleanup. + +### Environment variable filtering + +The local sandbox filters sensitive environment variables before passing them to commands. Variables ending in `_API_KEY`, `_SECRET`, `_TOKEN`, `_PASSWORD`, or `_CREDENTIAL` are stripped. A safelist of common variables (`PATH`, `HOME`, `USER`, `SHELL`, `LANG`, `TERM`, `TMPDIR`, `GOPATH`, `CARGO_HOME`, `NVM_DIR`) is always passed through. + + +The local sandbox offers no isolation. Agents can read and modify any file on the host. Use `docker` or `daytona` when running untrusted workflows or when you need a reproducible environment. + + +## Docker + +The Docker sandbox runs all tool operations inside a Docker container. The host working directory is bind-mounted into the container, so file changes are visible on both sides. + +### Prerequisites + +- Docker Engine running on the host +- The configured image available locally (or `auto_pull` enabled) + +### How it works + +- **Container lifecycle** — On `initialize()`, Arc pulls the image (if needed), creates a container with `sleep infinity`, and starts it. On `cleanup()`, Arc stops and removes the container. +- **Working directory** — The host working directory is bind-mounted at `/workspace` inside the container. All relative paths resolve against this mount point. +- **Commands** — Executed via `docker exec` with `/bin/bash -c` inside the container. Timeout and cancellation are supported. +- **File writes** — Use the Docker API's tar upload to avoid shell escaping issues with special characters. +- **Platform detection** — The container's `uname -r` is cached at startup. + +### Configuration + +The Docker sandbox is configured through the `DockerSandboxConfig`: + +| Setting | Default | Description | +|---|---|---| +| `image` | `arc-agent:latest` | Docker image to use | +| `container_mount_point` | `/workspace` | Mount point inside the container | +| `network_mode` | `bridge` | Docker network mode | +| `extra_mounts` | `[]` | Additional `host:container` bind mounts | +| `memory_limit` | unlimited | Memory limit in bytes | +| `cpu_quota` | unlimited | CPU quota (microseconds per 100ms period) | +| `auto_pull` | `true` | Pull the image if not found locally | +| `env_vars` | `[]` | Additional `KEY=VALUE` environment variables | + +### Preserving the container + +By default, the container is destroyed when the run finishes. To keep it alive for debugging: + +```bash +arc run start workflow.dot --sandbox docker --preserve-sandbox +``` + +Or in the run config: + +```toml +[sandbox] +provider = "docker" +preserve = true +``` + +When preserved, Arc prints the container ID so you can reconnect with `docker exec -it bash`. + +## Daytona + +The Daytona sandbox runs all tool operations inside a cloud-hosted VM managed by [Daytona](https://daytona.io). It provides full machine-level isolation, automatic git cloning, and SSH access for debugging. + +### Prerequisites + +- A `DAYTONA_API_KEY` environment variable +- The `gh` CLI authenticated (for git clone authentication) + +### How it works + +- **Sandbox lifecycle** — On `initialize()`, Arc creates a Daytona sandbox (from an image or a snapshot), clones the current git repository into it, and waits until it's ready. On `cleanup()`, the sandbox is deleted. +- **Working directory** — Fixed at `/home/daytona/workspace`. The current repository is cloned there automatically. +- **Git clone** — Arc detects the local `origin` remote URL and current branch, converts SSH URLs to HTTPS, obtains a GitHub token via `gh auth token`, and clones into the sandbox. If no git repo is detected, the working directory is created empty. +- **Commands** — Executed via the Daytona process API. Commands are base64-encoded and piped through `sh` to support pipes, environment variables, and other shell features. +- **Ephemeral** — Sandboxes are created with `ephemeral: true` and a unique timestamped name (e.g. `arc-20260305-142301-a3f2`). + +### Snapshots + +Snapshots let you pre-build an environment image so each run starts with dependencies already installed. If the named snapshot doesn't exist and a `dockerfile` is provided, Arc creates it automatically and polls until it's ready (up to 10 minutes). + +```toml +[sandbox] +provider = "daytona" + +[sandbox.daytona] +auto_stop_interval = 60 + +[sandbox.daytona.snapshot] +name = "rust-dev" +cpu = 4 +memory = 8 +disk = 20 +dockerfile = "FROM rust:1.85-slim-bookworm\nRUN apt-get update && apt-get install -y git ripgrep" +``` + +| Field | Description | +|---|---| +| `name` | Snapshot identifier. Reused across runs if it already exists. | +| `cpu` | CPU cores for the snapshot VM. | +| `memory` | Memory in GB. | +| `disk` | Disk in GB. | +| `dockerfile` | Dockerfile content for building the snapshot. Required when creating a new snapshot. | + +If the snapshot already exists and is in `Active` state, Arc uses it directly. If it's in `Building` or `Pending` state, Arc polls with exponential backoff until it's ready. + +### Labels + +Attach key-value labels to sandboxes for filtering and identification in the Daytona dashboard: + +```toml +[sandbox.daytona.labels] +project = "arc" +env = "ci" +team = "platform" +``` + +When using server defaults, labels are merged — run config labels override default labels on key collisions. + +### SSH access + +Connect to a running Daytona sandbox via SSH for live debugging: + +```bash +arc run start workflow.dot --sandbox daytona --ssh +``` + +This creates temporary SSH credentials (valid for 60 minutes) and prints the connection command. + +### Preserving the sandbox + +Like Docker, Daytona sandboxes are destroyed on cleanup by default. Use `--preserve-sandbox` to keep them alive: + +```bash +arc run start workflow.dot --sandbox daytona --preserve-sandbox +``` + +Arc prints the sandbox name so you can find it in the [Daytona dashboard](https://app.daytona.io/dashboard/sandboxes). + +### Auto-stop + +The `auto_stop_interval` setting (in minutes) tells Daytona to stop the sandbox after a period of inactivity. This saves costs for long-running sandboxes that may sit idle: + +```toml +[sandbox.daytona] +auto_stop_interval = 30 +``` + +### Network access + +Control outbound network access from the sandbox with the `network` setting. There are three modes: + +| Mode | Description | +|---|---| +| `"allow_all"` | Full internet access. This is the Daytona default when `network` is not set. | +| `"block"` | Block all egress. The sandbox cannot make any outbound connections. | +| `{ allow_list = [...] }` | Allow egress only to the listed CIDR ranges. | + +```toml +# Block all egress +[sandbox.daytona] +network = "block" + +# Full access (explicit) +[sandbox.daytona] +network = "allow_all" + +# Allow only specific CIDRs +[sandbox.daytona] +network = { allow_list = ["208.80.154.232/32", "10.0.0.0/8"] } +``` + +When using server defaults, the run config `network` overrides the server default. If neither specifies `network`, Daytona's own default (full access) applies. + +## Safety guardrails + +Regardless of which provider you use, Arc applies a **read-before-write** guardrail. Agents must read a file (via `read_file` or `grep`) before they can modify it with `write_file` or `delete_file`. Writing to new files that don't yet exist is always allowed. This prevents agents from blindly overwriting files they haven't inspected. diff --git a/docs/execution/failures.mdx b/docs/execution/failures.mdx index 26daec02e..9d8ef3d48 100644 --- a/docs/execution/failures.mdx +++ b/docs/execution/failures.mdx @@ -1,9 +1,265 @@ --- title: "Failures" -description: "Handling failures during workflow execution" +description: "How Arc classifies, retries, and recovers from failures during workflow execution" --- -## Model Failures +Failures are inevitable when orchestrating LLM-powered workflows — models hit rate limits, agents produce bad output, commands fail, and providers go down. Arc handles this with multiple layers of defense: automatic retries with backoff, provider failover, failure classification for intelligent routing, circuit breakers to prevent infinite loops, and goal gates to catch quality problems before a run completes. -## Loop Detection +## Failure classification +When a node fails, Arc classifies the failure into one of six categories. These classes drive retry decisions, circuit breaker logic, and edge routing. + +| Class | Description | Examples | +|---|---|---| +| `transient_infra` | Temporary infrastructure problem — likely to resolve on retry | Rate limits, timeouts, network errors, 5xx responses | +| `deterministic` | Permanent failure — retrying won't help | Authentication errors, bad configuration, invalid requests | +| `budget_exhausted` | Resource limit reached | Context length exceeded, token/turn limits, quota exhausted | +| `compilation_loop` | Reserved for loop detection | — | +| `canceled` | User or system cancellation | Cancel signal, abort | +| `structural` | Reserved for scope enforcement | Write scope violations | + +Classification happens automatically. Arc inspects SDK error types, HTTP status codes, and error message patterns to assign the right class. The `failure_class` is written to [context](/execution/context) after each stage, so you can route on it in edge conditions: + +```dot +implement -> fix [condition="failure_class=transient_infra"] +implement -> escalate [condition="failure_class=deterministic"] +``` + +## Retry layers + +Arc retries failures at two levels: **LLM retries** handle transient API errors inside a single model call, and **node retries** re-execute the entire node handler when the first level isn't enough. These layers are independent — a node retry re-runs the full handler, which gets its own fresh set of LLM retries. + +### LLM retries + +Every LLM call (within an agent session or a one-shot prompt node) has a built-in retry loop for transient API errors. This is invisible to the workflow — it happens inside the model call itself. + +| Setting | Default | +|---|---| +| Max retries | 3 | +| Initial delay | 1 second | +| Backoff multiplier | 2x | +| Max delay | 60 seconds | +| Jitter | 0.5x–1.5x random factor | + +Only transient errors are retried: rate limits, server errors (5xx), timeouts, network failures, and stream interruptions. Permanent errors like authentication failures or invalid requests fail immediately. + +If the provider returns a `Retry-After` header, Arc respects it — unless the delay exceeds 60 seconds, in which case the call fails rather than blocking the run. + +### Node retries + +When a node handler fails (after LLM retries are exhausted), the engine can retry the entire node. This is controlled by **retry policies**. + +#### Retry policies + +Set a retry policy on a node with the `retry_policy` attribute: + +```dot +implement [retry_policy="standard"] +``` + +| Policy | Max attempts | Initial delay | Backoff | Typical delays | +|---|---|---|---|---| +| `none` | 1 | — | — | No retries | +| `standard` | 5 | 200ms | 2x exponential | 200ms, 400ms, 800ms, 1.6s | +| `aggressive` | 5 | 500ms | 2x exponential | 500ms, 1s, 2s, 4s | +| `linear` | 3 | 500ms | 1x (constant) | 500ms, 500ms | +| `patient` | 3 | 2s | 3x exponential | 2s, 6s | + +All policies apply random jitter (0.5x–1.5x) and cap individual delays at 60 seconds. + +#### Setting retries without a policy + +You can also set just the retry count using `max_retries`: + +```dot +implement [max_retries="5"] +``` + +This uses the default backoff (200ms initial, 2x exponential) with the specified number of retries. + +#### Resolution order + +The engine resolves retry configuration in this order: + +1. Node attribute `retry_policy` — named preset +2. Node attribute `max_retries` — count only, default backoff +3. Graph attribute `default_max_retry` — applies to all nodes without explicit config (default: **3**) + +#### What gets retried + +Not all errors trigger a node retry. The handler's `should_retry` check must return true — generally, only errors classified as transient are retried. Deterministic errors (auth failures, bad config) fail immediately without consuming retry attempts. + +When a handler returns a `Retry` status instead of `Fail`, retries always proceed (if attempts remain). If retries are exhausted and the node has `allow_partial=true`, the outcome is promoted to `PartialSuccess` instead of failing. + +## Model fallbacks + +When a model provider fails with a transient error or quota exhaustion, Arc can automatically switch to a different provider. Configure fallback chains in your [run configuration](/execution/run-configuration): + +```toml +[llm] +model = "claude-opus-4-6" +provider = "anthropic" + +[llm.fallbacks] +anthropic = ["gemini", "openai"] +gemini = ["anthropic", "openai"] +``` + +When Anthropic is unavailable, Arc tries Gemini first, then OpenAI. For each fallback provider, Arc selects the closest model by matching required capabilities (tool use, vision, reasoning) and minimizing cost difference. + +### What triggers failover + +Failover is a superset of LLM retry eligibility: + +| Error type | LLM retry | Provider failover | +|---|---|---| +| Rate limit | Yes | Yes | +| Server error (5xx) | Yes | Yes | +| Timeout / network | Yes | Yes | +| Quota exceeded | No | Yes | +| Authentication (401) | No | No | +| Invalid request (400) | No | No | +| Context length (413) | No | No | +| Content filter | No | No | + +Quota errors are the key distinction — they aren't retried against the same provider (the quota won't reset) but *are* eligible for failover to a provider with its own quota. + +## Loop detection + +Arc has two independent mechanisms for detecting stuck loops: **node visit limits** that catch workflow-level cycles, and **tool call pattern detection** that catches agent-level repetition. + +### Node visit limits + +The `max_node_visits` graph attribute sets the maximum number of times any single node can execute before the run is terminated: + +```dot +digraph Example { + graph [max_node_visits="20"] + // ... +} +``` + +| Context | Default | +|---|---| +| Normal runs | Disabled (unlimited) | +| Dry runs (`--dry-run`) | 10 | +| Explicit `max_node_visits` | The configured value | + +When a node hits the limit, the run fails immediately: + +``` +node "verify" visited 20 times (limit 20); run is stuck in a cycle +``` + +### Tool call loop detection + +Inside an agent session, Arc monitors the last 10 assistant turns for repeating tool call patterns. It detects patterns of length 1 (same call repeated), 2 (A-B-A-B), or 3 (A-B-C-A-B-C). Every complete group in the window must match for detection to trigger. + +When a loop is detected, Arc injects a steering message into the conversation: + +> WARNING: Loop detected. You appear to be repeating the same tool calls. Please try a different approach or ask for clarification. + +This gives the agent a chance to break out of the loop without failing the node. + +## Failure signatures and circuit breakers + +Failure signatures are a deduplication mechanism that prevents the same failure from recurring indefinitely across loop iterations. They are particularly important for workflows with retry loops (implement → verify → fix → verify → ...). + +### How signatures work + +After each failed node, Arc constructs a **failure signature** — a normalized fingerprint combining the node ID, failure class, and error message: + +``` +implement|deterministic|handler panicked: index out of bounds +``` + +The error message is normalized by lowercasing, replacing hex strings with ``, replacing digits with ``, and truncating to 240 characters. This groups failures with the same root cause even when details like line numbers or timestamps vary. + +### Circuit breaker + +Arc tracks signature counts across the run. When the same signature repeats **3 times** (configurable via `loop_restart_signature_limit`), the run is terminated: + +``` +deterministic failure cycle detected: signature ... repeated 3 times (limit 3) +``` + +Only `deterministic` and `structural` failures are tracked — transient failures are excluded because they may genuinely resolve on retry. + + +Failure signature counts are never reset on success. This is intentional — it prevents cycles like "implement succeeds → verify fails → fix → implement succeeds → verify fails" from running indefinitely. + + +### Loop restart edges + +Edges marked with `loop_restart=true` trigger a special restart of the workflow from the target node. These have an additional guard: only `transient_infra` failures may cross a `loop_restart` edge. If the failure class is anything else, the run is terminated: + +``` +loop_restart blocked: failure_class=deterministic (requires transient_infra) +``` + +Loop restart edges also have their own separate circuit breaker (`restart_failure_signatures`) that enforces the same signature limit. + +## Goal gates + +Goal gates are quality checkpoints that are enforced when the workflow reaches an exit node. A node marked with `goal_gate=true` must have completed with `success` or `partial_success` — otherwise the run cannot finish. + +```dot +verify [shape=box, goal_gate="true"] +``` + +When a goal gate is unsatisfied at the exit node, Arc looks for a **retry target** — a node to jump back to for another attempt: + +1. Failed node's `retry_target` attribute +2. Failed node's `fallback_retry_target` attribute +3. Graph-level `retry_target` attribute +4. Graph-level `fallback_retry_target` attribute + +```dot +digraph Example { + graph [retry_target="plan"] + verify [shape=box, goal_gate="true", retry_target="implement"] + // If verify fails, jump to implement + // If no node-level target existed, would jump to plan +} +``` + +If no retry target is found at any level, the run fails: + +``` +goal gate unsatisfied for node verify and no retry target +``` + +## Stall watchdog + +Arc runs a background watchdog that monitors event activity. If no events are emitted for longer than the **stall timeout**, the run is canceled. This catches cases where a handler hangs indefinitely without producing errors. + +| Setting | Default | +|---|---| +| `stall_timeout` | 600 seconds (10 minutes) | +| Set to `0` | Disables the watchdog | + +```dot +digraph Example { + graph [stall_timeout="300"] // 5 minutes +} +``` + +## When failures become fatal + +A node failure does **not** automatically terminate the run. Arc follows this escalation path: + +1. **LLM retries** — transient API errors are retried inside the model call (up to 3 retries) +2. **Provider failover** — if configured, switch to a fallback provider +3. **Node retries** — re-execute the entire handler (per the retry policy) +4. **Edge routing** — if the node ultimately fails, look for an outgoing edge that matches (e.g., `condition="outcome=fail"`) +5. **Retry target** — if no matching edge exists, check `retry_target` / `fallback_retry_target` on the node and graph +6. **Run failure** — if none of the above produces a path forward, the run terminates + +The run also terminates immediately for: + +- **Node visit limit exceeded** — a node has been visited too many times +- **Circuit breaker tripped** — the same failure signature has repeated too many times +- **Loop restart blocked** — a non-transient failure tried to cross a `loop_restart` edge +- **Goal gate failure with no retry target** — a required gate was unsatisfied at the exit node +- **Stall timeout** — no events for too long +- **Cancellation** — user or system cancel signal diff --git a/docs/execution/interviews.mdx b/docs/execution/interviews.mdx index a52703fc1..8689678b4 100644 --- a/docs/execution/interviews.mdx +++ b/docs/execution/interviews.mdx @@ -1,9 +1,158 @@ --- title: "Interviews" -description: "Collecting information from users during workflow execution" +description: "How Arc collects human input during workflow execution" --- -## Question Types +When a workflow reaches a [human gate](/workflows/human-in-the-loop), it needs to pause and wait for a person to respond. The **interviewer** is the abstraction that makes this work — it presents a question, collects an answer, and returns it to the engine so execution can continue. + +Arc ships with several interviewer implementations for different environments: an interactive terminal prompt for the CLI, a web-based queue for the API server and web UI, an auto-approve mode for CI, and record/replay support for testing. + +## Question types + +Every human interaction is modeled as a `Question` with a type that determines how it's presented: + +| Type | Description | CLI presentation | +|---|---|---| +| `YesNo` | Binary yes/no decision | `[Y/N]` prompt | +| `Confirmation` | Confirm an action (like YesNo) | `[Y/N]` prompt | +| `MultipleChoice` | Pick one option from a list | Arrow-key selector or numbered list | +| `MultiSelect` | Pick one or more from a list | Checkbox selector | +| `Freeform` | Open-ended text input | `>` prompt | + +### Question structure + +Each question carries metadata beyond the prompt text: + +| Field | Description | +|---|---| +| `text` | The question displayed to the user | +| `question_type` | One of the types above | +| `options` | List of `{key, label}` pairs for choice questions | +| `allow_freeform` | Whether free-text input is accepted in addition to fixed options | +| `default` | Default answer used on timeout | +| `timeout_seconds` | How long to wait before using the default or timing out | +| `stage` | The node ID that generated this question | +| `metadata` | Arbitrary key-value metadata for integrations | + +## Answer values + +Answers are one of six variants: + +| Value | Meaning | +|---|---| +| `Yes` | Affirmative response to a yes/no or confirmation question | +| `No` | Negative response | +| `Selected(key)` | A specific option was chosen (carries the option key) | +| `Text(string)` | Free-text input | +| `Skipped` | The user dismissed the question without answering | +| `Timeout` | The question's timeout elapsed without a response | + +An answer can also carry a `selected_option` (the full `{key, label}` pair) and a `text` field for freeform input. + +## How human gates build questions + +When the engine reaches a human gate node (`shape=hexagon`), the [human handler](/workflows/human-in-the-loop) builds a question from the node's outgoing edges: + +1. Each edge becomes an option, with the accelerator key parsed from the label (e.g. `[A] Approve` → key `A`, label `[A] Approve`) +2. Edges with `freeform=true` are excluded from the option list and enable free-text fallback +3. The question text comes from the node's `label` attribute + +The handler then passes the question to the interviewer, waits for an answer, and maps it back to an edge for [transition](/workflows/transitions#human-gate-transitions). ## Channels +The `Interviewer` trait has a simple interface — `ask(question) → answer` — and Arc provides implementations for each delivery channel: + +### Console + +The default for CLI runs. On a TTY, the console interviewer uses interactive widgets (arrow-key selection, checkbox multi-select, confirm prompts) via `dialoguer`. When stdin is piped (non-TTY), it falls back to a line-based reader with numbered options. + +```bash +arc run start workflow.dot +# At a human gate: +# ? Approve Plan +# [1] A - [A] Approve +# [2] R - [R] Revise +# Select: +``` + +### Web + +The default for API server runs. The web interviewer holds questions in a queue until answers are submitted externally — typically by the web UI or a REST API call. Each question gets a unique ID (e.g. `q-1`), and the `ask()` call blocks on a oneshot channel until `submit_answer(id, answer)` is called. + +This decoupling means the workflow engine and the user interface can run in different processes. The web UI polls for pending questions and posts answers back to the API. + +### Slack + +Arc's Slack integration uses the web interviewer under the hood. When a human gate fires, the pending question is rendered as a Slack message with interactive buttons. When a user clicks a button, the Slack event handler calls `submit_answer()` on the web interviewer, unblocking the workflow. + +### Auto-approve + +For fully automated runs or CI pipelines, the auto-approve interviewer answers every question without human input: + +- `YesNo` / `Confirmation` → `Yes` +- `MultipleChoice` / `MultiSelect` → first option +- `Freeform` → `"auto-approved"` + +Enable it with the `--auto-approve` flag: + +```bash +arc run start workflow.dot --auto-approve +``` + +### Callback + +A programmatic interviewer that delegates to a provided function. Useful for testing or building custom integrations: + +```rust +let interviewer = CallbackInterviewer::new(|question| { + if question.question_type == QuestionType::YesNo { + Answer::yes() + } else { + Answer::text("custom response") + } +}); +``` + +### Queue + +A lightweight interviewer that pops answers from a pre-filled queue in order. Returns `Skipped` when the queue is empty. Useful for deterministic testing. + +## Timeouts + +Questions can have a `timeout_seconds` field. When set, Arc wraps the interviewer call with a timeout: + +- If the user answers before the deadline, their answer is used normally +- If the timeout elapses and a `default` answer is set on the question, the default is used +- If the timeout elapses with no default, the answer is `Timeout` + +The human handler then checks the node's `human.default_choice` attribute. If set, execution continues to the default target. Otherwise, the stage retries. + +```dot +approve [shape=hexagon, label="Approve?", human.default_choice="deploy"] +``` + +## Recording and replay + +The recording interviewer wraps any other interviewer and captures every question-answer pair. Recordings can be serialized to JSON and saved to a file: + +```rust +let recorder = RecordingInterviewer::new(Box::new(ConsoleInterviewer::new(styles))); + +// ... run workflow, recorder captures all interactions ... + +recorder.save_to_file(Path::new("recordings.json"))?; +``` + +The replay interviewer plays back recorded answers in sequence, ignoring the actual question text. When recordings are exhausted, it returns `Skipped`. This enables deterministic re-execution of workflows that originally required human input: + +```rust +let recordings = RecordingInterviewer::load_from_file(Path::new("recordings.json"))?; +let replayer = ReplayInterviewer::new(recordings); +``` + +Together, record and replay make it possible to: + +- **Test workflows** that include human gates without manual interaction +- **Reproduce runs** by replaying the exact same human decisions +- **Audit decisions** by inspecting the recorded question-answer pairs diff --git a/docs/execution/insights.mdx b/docs/execution/observability.mdx similarity index 80% rename from docs/execution/insights.mdx rename to docs/execution/observability.mdx index e6c335409..5a27f9fe7 100644 --- a/docs/execution/insights.mdx +++ b/docs/execution/observability.mdx @@ -1,5 +1,5 @@ --- -title: "Insights" +title: "Observability" description: "Observability and analysis of workflow runs" --- diff --git a/docs/execution/retros.mdx b/docs/execution/retros.mdx new file mode 100644 index 000000000..b37680c95 --- /dev/null +++ b/docs/execution/retros.mdx @@ -0,0 +1,186 @@ +--- +title: "Retros" +description: "Automatic retrospectives that analyze every workflow run" +--- + +After every workflow run, Arc generates a **retro** — a structured retrospective that captures what happened, what went well, and what didn't. Retros combine deterministic metrics extracted from the run's checkpoint with a qualitative narrative produced by an LLM agent that analyzes the full event stream. + +The goal is continuous improvement. Retros give you a searchable history of how your workflows perform over time, surface friction patterns that would otherwise go unnoticed, and identify follow-up work before it falls through the cracks. + +## What's in a retro + +A retro has two layers: **quantitative stats** derived cheaply from checkpoint data, and an **agent-generated narrative** that interprets the run holistically. + +### Quantitative layer + +The quantitative layer is extracted directly from the [checkpoint](/execution/checkpoints) and event stream — no LLM calls required: + +| Field | Description | +|---|---| +| **Per-stage breakdown** | Duration, retry count, cost, files touched, status, and failure reason for each stage | +| **Aggregate stats** | Total duration, total cost, total retries, all files touched, stages completed vs. failed | + +### Narrative layer + +An LLM agent reads the run's `progress.jsonl` event stream and produces a structured analysis: + +| Field | Description | +|---|---| +| **Smoothness** | Overall rating on a 5-point scale (see below) | +| **Intent** | What the run was trying to accomplish | +| **Outcome** | What actually happened | +| **Learnings** | What was discovered about the repo, code, workflow, or tools | +| **Friction points** | Where things got stuck and why | +| **Open items** | Follow-up work, tech debt, test gaps, or investigations identified | + +The agent has tool access to grep and read the event stream, so it can inspect actual tool call patterns, error messages, and approach pivots — not just pass/fail signals. + +## Smoothness ratings + +Every retro includes a smoothness rating that grades the overall quality of the run's execution: + +| Rating | Meaning | +|---|---| +| **Effortless** | Goal achieved on the first try. No retries, no wrong approaches. Agent moved efficiently from start to finish. | +| **Smooth** | Goal achieved with minor hiccups — 1–2 retries or a brief wrong approach quickly corrected. No human intervention needed. | +| **Bumpy** | Goal achieved but with notable friction: multiple retries, at least one significant wrong approach, or substantial time on dead ends. | +| **Struggled** | Goal achieved only with difficulty: many retries, major approach changes, human intervention, or partial failures requiring recovery. | +| **Failed** | Run did not achieve its stated goal. Some stages may have completed, but the overall intent was not fulfilled. | + +The rating considers the full context visible in agent events — tool call patterns, error recovery sequences, approach pivots — not just stage pass/fail counts. + +## Learnings, friction points, and open items + +### Learnings + +Learnings capture what was discovered during the run, categorized by type: + +| Category | Examples | +|---|---| +| `repo` | Repository structure, build system quirks, CI configuration | +| `code` | Bug root causes, module boundaries, API contracts | +| `workflow` | Node ordering issues, missing stages, prompt improvements | +| `tool` | Tool limitations, MCP server behavior, command output parsing | + +### Friction points + +Friction points identify where the run got stuck and what caused the slowdown: + +| Kind | Description | +|---|---| +| `retry` | A stage needed multiple attempts | +| `timeout` | A stage or tool call hit a time limit | +| `wrong_approach` | The agent pursued a dead end before pivoting | +| `tool_failure` | A tool or command failed unexpectedly | +| `ambiguity` | Unclear requirements or conflicting signals caused confusion | + +Each friction point can optionally reference the `stage_id` where it occurred. + +### Open items + +Open items capture follow-up work identified during the run: + +| Kind | Description | +|---|---| +| `tech_debt` | Code quality issues worth addressing later | +| `follow_up` | Work that's related but out of scope for this run | +| `investigation` | Unknowns that need further research | +| `test_gap` | Missing test coverage discovered during the run | + +## How retros are generated + +Retro generation happens in two phases after a run completes: + +1. **Derive** — Arc extracts stage durations from `progress.jsonl` and builds a retro from the checkpoint data. This is deterministic, fast, and produces the quantitative layer. The retro is saved immediately as `retro.json` in the run's logs directory. + +2. **Narrate** — An LLM agent session analyzes the run data. The agent has read access to `progress.jsonl`, `checkpoint.json`, and `manifest.json`. It uses grep and read tools to find interesting signals — failures, retries, errors, approach changes — then calls a `submit_retro` tool with its structured analysis. The narrative fields are merged into the existing retro and saved. + +Both phases run automatically at the end of every CLI run. The API server derives the quantitative layer but does not currently run the narrative agent. + +## Accessing retros + +### CLI + +Retros are saved to `{logs_dir}/retro.json` after every run. The path is printed at the end of the run output: + +``` +Retro: smooth — Successfully implemented the feature + Retro saved to ~/arc-logs/01JKXYZ.../retro.json +``` + +To skip retro generation, pass `--no-retro`: + +```bash +arc run start workflow.dot --no-retro +``` + +### API + +Retrieve the retro for a specific run: + +``` +GET /runs/{id}/retro +``` + +Returns the full retro JSON, or `null` if no retro is available yet. + +List retros across all runs (paginated): + +``` +GET /retros +``` + +Returns a summary for each retro including `run_id`, `workflow_name`, `goal`, `timestamp`, `smoothness`, aggregate stats, and a count of friction points. + +## Storage + +Retros are stored as `retro.json` in the run's logs directory alongside `checkpoint.json` and `progress.jsonl`. They are plain JSON files — easy to parse, query, or pipe into other tools. + +```json +{ + "run_id": "01JKXYZ...", + "workflow_name": "PlanImplement", + "goal": "Fix the authentication bug", + "timestamp": "2026-03-05T14:30:00Z", + "smoothness": "smooth", + "stages": [ + { + "stage_id": "plan", + "stage_label": "Plan", + "status": "success", + "duration_ms": 5000, + "retries": 0, + "cost": 0.05, + "files_touched": ["plan.md"] + }, + { + "stage_id": "implement", + "stage_label": "Implement", + "status": "success", + "duration_ms": 15000, + "retries": 1, + "cost": 0.10, + "files_touched": ["src/auth.rs", "src/auth_test.rs"] + } + ], + "stats": { + "total_duration_ms": 20000, + "total_cost": 0.15, + "total_retries": 1, + "files_touched": ["plan.md", "src/auth.rs", "src/auth_test.rs"], + "stages_completed": 2, + "stages_failed": 0 + }, + "intent": "Fix the authentication bug causing login failures", + "outcome": "Successfully fixed the token refresh logic", + "learnings": [ + { "category": "code", "text": "Token refresh was in the wrong module" } + ], + "friction_points": [ + { "kind": "retry", "description": "First implementation attempt had an incorrect import", "stage_id": "implement" } + ], + "open_items": [ + { "kind": "test_gap", "description": "No integration test for token refresh flow" } + ] +} +``` diff --git a/docs/execution/run-configuration.mdx b/docs/execution/run-configuration.mdx index 3ee21ab30..f8f945ad1 100644 --- a/docs/execution/run-configuration.mdx +++ b/docs/execution/run-configuration.mdx @@ -1,5 +1,292 @@ --- title: "Run Configuration" -description: "Configuring workflow runs" +description: "Configure workflow runs with TOML files" --- +A run config is a TOML file that bundles a workflow graph with all the settings needed to execute it — the goal, model, sandbox, setup commands, variables, and hooks. Instead of passing a dozen CLI flags, you check a `.toml` file into version control and launch with a single command: + +```bash +arc run start run.toml +``` + +## Minimal example + +A run config requires three fields: + +```toml +version = 1 +goal = "Implement the login feature" +graph = "workflow.dot" +``` + +| Field | Description | +|---|---| +| `version` | Config format version. Must be `1`. | +| `goal` | What the workflow should accomplish. Passed to agents and used in retrospectives. | +| `graph` | Path to the DOT workflow file, resolved relative to the TOML file's directory. | + +## Full example + +```toml +version = 1 +goal = "Run the CI pipeline for $repo_name" +graph = "pipelines/ci.dot" +directory = "/tmp/workdir" + +[llm] +model = "claude-sonnet-4-5" +provider = "anthropic" + +[llm.fallbacks] +anthropic = ["gemini", "openai"] +gemini = ["anthropic", "openai"] + +[setup] +commands = ["git clone $repo_url repo", "cd repo && npm install"] +timeout_ms = 120000 + +[sandbox] +provider = "daytona" +preserve = false + +[sandbox.daytona] +auto_stop_interval = 60 + +[sandbox.daytona.labels] +project = "arc" +env = "ci" + +[sandbox.daytona.snapshot] +name = "node-20" +cpu = 4 +memory = 8 +disk = 20 +dockerfile = "FROM node:20-slim\nRUN apt-get update && apt-get install -y git" + +[vars] +repo_name = "arc" +repo_url = "https://github.com/qltysh/arc" + +[[hooks]] +event = "stage_start" +command = "./scripts/pre-check.sh" +blocking = true +sandbox = false + +[[hooks]] +event = "run_complete" +command = "echo done" +``` + +## Sections + +### `[llm]` + +Override the default model and provider for all nodes that don't have an explicit model assigned via a [stylesheet](/workflows/stylesheets). + +```toml +[llm] +model = "claude-sonnet-4-5" +provider = "anthropic" +``` + +| Field | Description | +|---|---| +| `model` | Model ID or alias (e.g. `claude-sonnet-4-5`, `opus`, `gemini-pro`). See [Models](/core-concepts/models). | +| `provider` | Provider name: `anthropic`, `openai`, `gemini`, `kimi`, `zai`, `minimax`, `inception`. | + +#### `[llm.fallbacks]` + +Map each provider to an ordered list of fallback providers. When the primary provider is unavailable, Arc tries the fallbacks in order: + +```toml +[llm.fallbacks] +anthropic = ["gemini", "openai"] +gemini = ["anthropic", "openai"] +``` + +### `[setup]` + +Shell commands to run before the workflow starts. Use this to clone repositories, install dependencies, or prepare the environment. + +```toml +[setup] +commands = ["pip install -r requirements.txt", "npm install"] +timeout_ms = 60000 +``` + +| Field | Description | +|---|---| +| `commands` | List of shell commands, executed sequentially via `sh -c`. | +| `timeout_ms` | Per-command timeout in milliseconds. Default: `300000` (5 minutes). | + +Each command must exit with status 0. If any command fails or times out, the run aborts before the workflow starts. + +### `[sandbox]` + +Configure how agent tools (bash, file edits) are executed. + +```toml +[sandbox] +provider = "docker" +preserve = true +``` + +| Field | Description | +|---|---| +| `provider` | Sandbox mode: `local` (default), `docker`, or `daytona`. | +| `preserve` | When `true`, keep the sandbox alive after the run finishes. Useful for debugging. | + +#### `[sandbox.daytona]` + +Additional settings when using the Daytona cloud sandbox: + +```toml +[sandbox.daytona] +auto_stop_interval = 60 + +[sandbox.daytona.labels] +project = "arc" +env = "staging" + +[sandbox.daytona.snapshot] +name = "my-snapshot" +cpu = 4 +memory = 8 +disk = 20 +dockerfile = "FROM rust:1.85-slim-bookworm\nRUN apt-get update" +``` + +| Field | Description | +|---|---| +| `auto_stop_interval` | Minutes of inactivity before the sandbox auto-stops. | +| `labels` | Key-value labels attached to the sandbox for filtering and identification. | +| `snapshot.name` | Snapshot name to create or use for the sandbox. | +| `snapshot.cpu` | CPU cores for the snapshot. | +| `snapshot.memory` | Memory in GB for the snapshot. | +| `snapshot.disk` | Disk in GB for the snapshot. | +| `snapshot.dockerfile` | Dockerfile content for building the snapshot image. | +| `network` | Network access mode: `"allow_all"` (default), `"block"`, or `{ allow_list = ["..."] }`. See [Sandboxing](/administration/sandboxing#network-access-control). | + +### `[vars]` + +Define variables that are expanded into the DOT source before the graph is parsed. See [Variables](/workflows/variables) for the full reference. + +```toml +[vars] +repo_name = "arc" +repo_url = "https://github.com/qltysh/arc" +language = "rust" +``` + +Variables can be used anywhere in the DOT file with `$name` syntax: + +```dot +digraph CI { + graph [goal="Run tests for $repo_name"] + clone [shape=parallelogram, script="git clone $repo_url repo"] + test [label="Test", prompt="Run the $language test suite."] +} +``` + +If a `$variable` in the DOT file has no matching entry in `[vars]`, Arc raises an error immediately. A bare `$` not followed by an identifier (e.g. `costs $5`) is left as-is. + +### `[[hooks]]` + +Define hooks that run in response to lifecycle events. Each hook is a TOML array entry: + +```toml +[[hooks]] +name = "pre-check" +event = "stage_start" +command = "./scripts/pre-check.sh" +matcher = "agent_loop" +blocking = true +timeout_ms = 30000 +sandbox = false +``` + +| Field | Description | +|---|---| +| `name` | Optional display name for the hook. | +| `event` | Lifecycle event: `run_start`, `run_complete`, `stage_start`, `stage_complete`. | +| `command` | Shell command to execute (shorthand for `type = "command"`). | +| `matcher` | Regex matched against node ID or handler type. Limits which stages trigger this hook. | +| `blocking` | Whether the hook must complete before execution continues. Defaults vary by event. | +| `timeout_ms` | Hook timeout in milliseconds. Default: `60000` (60s). | +| `sandbox` | Run inside the sandbox (`true`, default) or on the host (`false`). | + +See [Hooks](/agents/hooks) for hook types beyond simple commands (HTTP, prompt, agent). + +## Top-level fields + +In addition to the sections above, two optional top-level fields are available: + +| Field | Description | +|---|---| +| `directory` | Working directory for the run. Defaults to the current directory. | + +## Graph path resolution + +The `graph` path is resolved relative to the TOML file's parent directory, not the current working directory. This means a run config and its workflow can live side by side: + +``` +project/ + runs/ + ci.toml # graph = "ci.dot" + ci.dot +``` + +Absolute paths are used as-is. + +## Precedence + +Settings can come from multiple sources. Arc resolves them in this order (first match wins): + +| Source | Priority | +|---|---| +| Node-level [stylesheet](/workflows/stylesheets) | Highest | +| Run config TOML | | +| CLI flags (`--model`, `--provider`, `--sandbox`) | | +| Server defaults (`~/.arc/server.toml`) | | +| DOT graph attributes (`default_model`, `default_provider`) | | +| Built-in defaults | Lowest | + + +For model and provider specifically, the precedence is: CLI flags > TOML config > server defaults > DOT graph attributes > built-in defaults. Stylesheet rules on individual nodes always take priority over all of these. + + +### Server defaults + +When running via `arc serve`, the server config at `~/.arc/server.toml` can set default values for `[llm]`, `[setup]`, `[sandbox]`, and `[vars]`. These defaults are applied to every run unless the run config overrides them. + +For variables, defaults and run config are **merged** — the run config wins on key collisions: + +```toml +# ~/.arc/server.toml +[vars] +default_key = "from_server" +shared = "from_server" + +# run.toml +[vars] +shared = "from_run" # wins +task_key = "from_run" +``` + +The same merge behavior applies to Daytona labels. All other fields use simple "first non-empty wins" precedence. + +## Validation + +Arc validates the run config when it loads: + +- **Version check** — Only `version = 1` is accepted. Other versions are rejected immediately. +- **Required fields** — `version`, `goal`, and `graph` are all required. Missing any of them is an error. +- **Unknown fields** — Extra fields not listed above are rejected (`deny_unknown_fields`). +- **Variable check** — Any `$variable` in the DOT file without a matching `[vars]` entry produces an error. + +Use `--preflight` to validate a run config without executing it: + +```bash +arc run start run.toml --preflight +``` diff --git a/docs/workflows/stages-and-nodes.mdx b/docs/workflows/stages-and-nodes.mdx index f16d6a72f..e9ca9aa21 100644 --- a/docs/workflows/stages-and-nodes.mdx +++ b/docs/workflows/stages-and-nodes.mdx @@ -50,7 +50,7 @@ Key attributes: | `prompt` | The task instructions for the agent | | `reasoning_effort` | `low`, `medium`, or `high` (default: `high`) | | `max_tokens` | Maximum tokens for LLM responses | -| `fidelity` | How much prior context is passed to this node (see table below) | +| `fidelity` | How much prior context is passed to this node (see [Context](/execution/context#fidelity-controlling-agent-context)) | | `thread_id` | Groups nodes into a shared conversation thread (advanced — see below) | | `timeout` | Execution timeout (e.g. `"900s"`) | @@ -65,7 +65,7 @@ Key attributes: | `summary:low` | Brief summary with just outcomes per stage | | `truncate` | Minimal — only the goal and run ID | -Fidelity can also be set at the graph level (`default_fidelity`) or on individual edges to control the transition between stages. +Fidelity can also be set at the graph level (`default_fidelity`) or on individual edges to control the transition between stages. See [Context](/execution/context#fidelity-controlling-agent-context) for the full reference on fidelity precedence, preamble construction, and thread integration. **Thread ID:**