diff --git a/docs/internal/events.md b/docs/internal/events.md deleted file mode 100644 index e90e8dc51..000000000 --- a/docs/internal/events.md +++ /dev/null @@ -1,2275 +0,0 @@ -# Events - -Every serialized run event envelope, whether streamed over SSE, returned by `fabro events`, or written to a JSONL sink, uses this structure: - -```json -{ - "id": "019234ab-cdef-7890-abcd-ef1234567890", - "ts": "2026-04-01T12:00:00.123Z", - "run_id": "01JQXYZ...", - "event": "stage.completed", - "session_id": "ses_abc", - "parent_session_id": "ses_parent", - "node_id": "code", - "node_label": "Write Code", - "properties": { ... } -} -``` - -### Envelope fields - -| Field | Type | Description | -|-------|------|-------------| -| `id` | string | UUID v7 (time-ordered), unique per event | -| `ts` | string | RFC 3339 timestamp with millisecond precision | -| `run_id` | string | ULID of the run | -| `event` | string | Dot-notation event name | -| `session_id` | string? | Agent session id (agent events only) | -| `parent_session_id` | string? | Parent agent session id (agent events only) | -| `node_id` | string? | Node id (stage, checkpoint, agent, parallel branch, and other node-scoped events) | -| `node_label` | string? | Display label for the node (defaults to `node_id` when not set separately) | -| `properties` | object | Event-specific fields | - ---- - -## Run events - -### `run.created` - -Emitted when the run record is created. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.created", - "properties": { - "workflow_slug": "my-workflow", - "workflow_version_id": "wv_...", - "target": { - "kind": "git", - "repo": "acme/my-project", - "branch": "main", - "sha": "0123456789abcdef0123456789abcdef01234567" - }, - "source_directory": "/home/user/src/my-project", - "git": { - "origin_url": "https://github.com/acme/my-project", - "branch": "main", - "sha": "0123456789abcdef0123456789abcdef01234567", - "dirty": "clean" - }, - "fork_source_ref": null, - "in_place": false, - "provenance": { - "subject": { - "kind": "user", - "identity": { - "issuer": "https://github.com", - "subject": "12345" - }, - "login": "octocat", - "auth_method": "github" - } - } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `settings` | object | Workflow settings snapshot | -| `graph` | object | Parsed workflow graph | -| `workflow_source` | string? | Workflow source text | -| `labels` | object | Run labels | -| `source_directory` | string? | Submitter-side source directory | -| `workflow_slug` | string? | Workflow slug | -| `workflow_version_id` | string? | Exact immutable root workflow version used for admission | -| `target` | object? | Canonical accepted workspace target. Version-backed Git intent runs persist `kind`, `repo`, required `branch`, and optional normalized `sha`; legacy manifest runs omit it | -| `provenance` | object | Actor and request provenance | -| `manifest_blob` | string? | Blob hash for the submitted manifest | -| `git` | object? | Operational Git projection: normalized `origin_url`, `branch`, optional `sha`, and `dirty` status. For Git intent runs, `branch` is the submitted working branch and `sha` is the optional lowercase-normalized submitted commit; admission does not resolve it or prove branch ancestry. Legacy runs retain their observed optional-SHA semantics | -| `fork_source_ref` | object? | Source run/checkpoint reference when this run was forked | -| `in_place` | boolean | Whether the run was created with `--in-place` (no git checkpoints) | - -Readers remain tolerant of the legacy `workflow_config`, `run_dir`, and -`db_prefix` properties, and of a legacy `push_outcome` object nested inside -`git`, when replaying historical events; newly emitted `run.created` events -omit them. - -### `run.started` - -Emitted when the workflow run begins. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.started", - "properties": { - "name": "my-workflow", - "base_branch": "main", - "base_sha": "abc123...", - "run_branch": "fabro/run-01JQXYZ", - "worktree_dir": "/tmp/fabro-worktrees/...", - "goal": "Fix the login bug" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `name` | string | Workflow name | -| `base_branch` | string? | Base git branch | -| `base_sha` | string? | Base commit SHA | -| `run_branch` | string? | Git branch created for this run | -| `worktree_dir` | string? | Worktree directory path | -| `goal` | string? | Workflow goal text | - -Note: `run_id` is in the envelope, not in properties. - -### `run.completed` - -Emitted when the workflow run finishes successfully (or with partial success). - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.completed", - "properties": { - "duration_ms": 45000, - "artifact_count": 3, - "status": "succeeded", - "final_git_commit_sha": "def456...", - "usage": { - "tokens": { - "input": 15000, - "output": 5000, - "reasoning": 2000, - "cache_read": 8000, - "cache_write": 3000 - }, - "cost": { "usd_micros": 150000, "source": "catalog" } - } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `duration_ms` | number | Total run duration in milliseconds | -| `artifact_count` | number | Number of artifacts produced | -| `status` | string | Final stage outcome (`"succeeded"`, `"failed"`, `"partially_succeeded"`, `"skipped"`) | -| `final_git_commit_sha` | string? | Final HEAD SHA | -| `usage` | object? | The run's usage summed across every stage visit, as lithos-llm's `Usage`. Absent for a run that made no model calls | -| `usage.tokens` | object | The five disjoint token buckets: `input`, `output`, `reasoning`, `cache_read`, `cache_write`. Their plain sum is the total | -| `usage.cost` | object? | `usd_micros` and `source` (`catalog`, `provider`, or `application`). Absent when the cost is unknown, never zero: a sum has a cost only when every part that used tokens was priced | - -### `run.failed` - -Emitted when the workflow run fails. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.failed", - "properties": { - "error": "Handler error: compilation failed", - "duration_ms": 12000, - "git_commit_sha": "abc123..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `error` | string | Error message (Display representation) | -| `duration_ms` | number | Run duration before failure | -| `git_commit_sha` | string? | HEAD SHA at time of failure | - -### `run.notice` - -Informational, warning, or error notice emitted during the run. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.notice", - "properties": { - "level": "warn", - "code": "missing_env_var", - "message": "GITHUB_TOKEN not set, PR creation will be skipped" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `level` | string | `"info"`, `"warn"`, or `"error"` | -| `code` | string | Machine-readable notice code | -| `message` | string | Human-readable message | - -### `run.interrupt` - -Emitted after a live worker accepts a run interrupt control operation. The -actor is stored in the top-level `actor` envelope field. Properties are empty. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.interrupt", - "actor": { "kind": "user", "login": "octocat" }, - "properties": {} -} -``` - -### `run.steer` - -Emitted after a live worker accepts run steering text. The actor is stored in -the top-level `actor` envelope field. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "run.steer", - "actor": { "kind": "user", "login": "octocat" }, - "properties": { - "text": "Remember to run tests after changes" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `text` | string | Accepted steering text | - -### `metadata.snapshot.started` - -Historical event, no longer emitted. Recorded when Fabro began a Git metadata snapshot operation. Retained for reading older run streams. - -Init and finalize metadata snapshots are unscoped. Checkpoint metadata snapshots use the checkpoint stage scope, so they include the checkpoint `node_id`, `node_label`, and `stage_id`. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "metadata.snapshot.started", - "properties": { - "phase": "checkpoint", - "branch": "fabro/meta" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `phase` | string | Logical metadata operation: `"init"`, `"checkpoint"`, or `"finalize"` | -| `branch` | string | Metadata branch/ref being written | - -### `metadata.snapshot.completed` - -Historical event, no longer emitted. Recorded when Fabro committed and pushed a metadata snapshot successfully. Retained for reading older run streams. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "metadata.snapshot.completed", - "properties": { - "phase": "checkpoint", - "branch": "fabro/meta", - "duration_ms": 2800, - "entry_count": 12, - "bytes": 18432, - "commit_sha": "def456..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `phase` | string | Logical metadata operation: `"init"`, `"checkpoint"`, or `"finalize"` | -| `branch` | string | Metadata branch/ref that was written | -| `duration_ms` | number | End-to-end duration of the metadata snapshot operation | -| `entry_count` | number | Number of metadata files written into the snapshot commit | -| `bytes` | number | Sum of serialized metadata entry byte lengths | -| `commit_sha` | string | Metadata snapshot commit SHA | - -### `metadata.snapshot.failed` - -Historical event, no longer emitted. Recorded when a metadata snapshot attempt failed, before the matching compatibility `run.notice`, allowing human-facing consumers to suppress duplicate warning text. Compatibility notices with codes `checkpoint_metadata_write_failed` and `checkpoint_metadata_push_failed` may still appear in raw event streams. The `checkpoint_metadata_degraded` notice is a separate summary signal and should not be treated as a duplicate of this event. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "metadata.snapshot.failed", - "properties": { - "phase": "checkpoint", - "branch": "fabro/meta", - "duration_ms": 900, - "failure_kind": "push", - "error": "failed to push metadata snapshot", - "causes": ["remote rejected the push"], - "commit_sha": "def456...", - "entry_count": 12, - "bytes": 18432 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `phase` | string | Logical metadata operation: `"init"`, `"checkpoint"`, or `"finalize"` | -| `branch` | string | Metadata branch/ref being written | -| `duration_ms` | number | End-to-end duration before failure | -| `failure_kind` | string | Failure phase: `"load_state"`, `"write"`, or `"push"` | -| `error` | string | Primary error summary | -| `causes` | string[] | Error cause chain; omitted when empty | -| `commit_sha` | string? | Local metadata commit SHA for push failures; omitted for load-state and write failures | -| `entry_count` | number? | Metadata entry count for push failures; omitted for load-state and write failures | -| `bytes` | number? | Serialized metadata byte count for push failures; omitted for load-state and write failures | - ---- - -## Stage events - -### `stage.started` - -Emitted when a workflow node begins execution. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "stage.started", - "node_id": "code", - "node_label": "Write Code", - "properties": { - "index": 1, - "handler_type": "agent", - "attempt": 1, - "max_attempts": 3 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Stage execution order index | -| `handler_type` | string | Handler type (`"agent"`, `"prompt"`, `"command"`, `"conditional"`, `"human"`, `"parallel"`, etc.) | -| `attempt` | number | Current attempt number (1-based) | -| `max_attempts` | number | Maximum attempts allowed | - -### `stage.completed` - -Emitted when a workflow node finishes execution. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "stage.completed", - "node_id": "code", - "node_label": "Write Code", - "properties": { - "index": 1, - "duration_ms": 8000, - "status": "succeeded", - "preferred_label": "tests_pass", - "suggested_next_ids": ["review"], - "usage": { - "model": { "provider": "anthropic", "model_id": "claude-sonnet-4-20250514" }, - "usage": { - "tokens": { - "input": 5000, - "output": 2000, - "reasoning": 500, - "cache_read": 3000, - "cache_write": 1000 - }, - "cost": { "usd_micros": 50000, "source": "catalog" } - } - }, - "usage_by_model": [], - "error": "lint failed", - "failure_class": "deterministic", - "failure_signature": "clippy::unused_import", - "context_updates": {"response.code": "done"}, - "jump_to_node": "review", - "context_values": {"response.code": "done"}, - "node_visits": {"code": 1}, - "loop_failure_signatures": {"code|deterministic|clippy::unused_import": 2}, - "restart_failure_signatures": {"code|transient_infra|timeout": 1}, - "response": "done", - "notes": "All tests passing", - "files_touched": ["src/main.rs", "src/lib.rs"], - "attempt": 1, - "max_attempts": 3 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Stage execution order index | -| `duration_ms` | number | Stage duration in milliseconds | -| `status` | string | `"succeeded"`, `"failed"`, `"skipped"`, `"partially_succeeded"` | -| `preferred_label` | string? | Edge label hint for routing | -| `suggested_next_ids` | string[] | Suggested successor node ids | -| `usage` | object? | The stage's usage under the model it ran on (`ModelUsage`): for an agent stage, the whole session tree's tokens under the root's route. Absent for a stage that made no model calls | -| `usage.model` | object | `provider`, `model_id`, and optional `speed` tier | -| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost sums what lithos-llm attached to each answer, the provider's reported figure when it gave one, else the catalog's price for the route; absent when an answer had neither | -| `usage_by_model` | array? | For an agent stage, `usage` split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know priced at the root's. Each row is a `ModelUsage`, and the rows sum to `usage`. Empty for stages without a coding agent and on events written before it existed | -| `error` | string? | Error message (flattened from failure detail) | -| `failure_class` | string? | `"transient_infra"`, `"deterministic"`, `"budget_exhausted"`, `"compilation_loop"`, `"canceled"`, `"structural"` | -| `failure_signature` | string? | Dedup key for repeated failures | -| `context_updates` | object? | Context delta written by this stage | -| `jump_to_node` | string? | Non-edge jump target | -| `context_values` | object? | Context snapshot after the stage, minus runtime-only keys such as `current.preamble`. Artifact pointers are not normalized to blob refs — use `checkpoint.completed` for the durable projection | -| `node_visits` | object? | Node visit counts after the stage | -| `loop_failure_signatures` | object? | Loop failure signature counts | -| `restart_failure_signatures` | object? | Restart failure signature counts | -| `response` | string? | Full LLM or agent response text when produced by the stage | -| `notes` | string? | Free-text notes | - -An agent stage's usage is its whole session tree's: the root session and -every subagent, live in `StageProjection.usage` and here at completion, both -read from the same fold of the stage's agent events. The root is priced at -its route and each subagent at its own model; where the provider reported a -cost, that cost stands. -| `files_touched` | string[] | File paths modified | -| `attempt` | number | Attempt number (1-based) | -| `max_attempts` | number | Maximum attempts allowed | - -Note: `failure` is flattened — the `failure.message` becomes `error`, `failure.failure_class` becomes `failure_class`, `failure.failure_signature` becomes `failure_signature`. - -### `stage.failed` - -Emitted when a stage fails (before retry decision). - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "stage.failed", - "node_id": "code", - "node_label": "Write Code", - "properties": { - "index": 1, - "error": "compilation failed", - "failure_class": "deterministic", - "failure_signature": "rustc::E0308", - "will_retry": true - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Stage execution order index | -| `error` | string | Error message (flattened from failure detail) | -| `failure_class` | string | Failure category | -| `failure_signature` | string? | Dedup key for repeated failures | -| `will_retry` | boolean | Whether the stage will be retried | -| `usage` | object? | What the stage spent before it failed, in the shape `stage.completed` uses. An agent stage that fails for good after answering model calls records its whole session tree, as it would have on completion; a retried attempt and a cancelled stage carry none | -| `usage_by_model` | array? | `usage` split by model, as on `stage.completed` | - -### `stage.retrying` - -Emitted when a stage is about to be retried. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "stage.retrying", - "node_id": "code", - "node_label": "Write Code", - "properties": { - "index": 1, - "attempt": 2, - "max_attempts": 3, - "delay_ms": 1000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Stage execution order index | -| `attempt` | number | Next attempt number | -| `max_attempts` | number | Maximum attempts allowed | -| `delay_ms` | number | Delay before retry in milliseconds | - -### `stage.prompt` - -Emitted when a prompt is rendered for an LLM stage. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "stage.prompt", - "node_id": "code", - "node_label": "code", - "properties": { - "text": "You are a coding agent. Fix the bug in..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `text` | string | Rendered prompt text | - ---- - -## Parallel events - -### `parallel.started` - -Emitted when a parallel node begins executing branches. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "parallel.started", - "properties": { - "visit": 1, - "branch_count": 3 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `visit` | number | Visit number for this parallel stage | -| `branch_count` | number | Number of parallel branches | - -### `parallel.branch.started` - -Emitted when a parallel branch begins. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "parallel.branch.started", - "node_id": "branch_a", - "node_label": "branch_a", - "properties": { - "index": 0 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Branch index | - -### `parallel.branch.completed` - -Emitted when a parallel branch finishes. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "parallel.branch.completed", - "node_id": "branch_a", - "node_label": "branch_a", - "properties": { - "index": 0, - "duration_ms": 5000, - "status": "succeeded" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `index` | number | Branch index | -| `duration_ms` | number | Branch duration in milliseconds | -| `status` | string | Branch outcome status | - -### `parallel.completed` - -Emitted when all parallel branches have finished. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "parallel.completed", - "properties": { - "visit": 1, - "duration_ms": 12000, - "success_count": 2, - "failure_count": 1, - "results": [ - { - "id": "branch_a", - "status": "succeeded", - "context_updates": {"response.branch_a": "review complete"} - }, - { - "id": "branch_b", - "status": "failed", - "context_updates": {"command.output": "validation failed"} - } - ] - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `visit` | number | Visit number for this parallel stage | -| `duration_ms` | number | Total parallel duration | -| `success_count` | number | Branches that succeeded | -| `failure_count` | number | Branches that failed | -| `results` | array | Ordered typed branch results with `id`, `status`, and isolated `context_updates` | - ---- - -## Interview events - -### `interview.started` - -Emitted when a human-in-the-loop question is posed. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "interview.started", - "node_id": "review", - "node_label": "review", - "properties": { - "question": "Does this look correct?", - "question_type": "approval" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `question` | string | Question text | -| `question_type` | string | Type of question | - -### `interview.completed` - -Emitted when a human answers. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "interview.completed", - "properties": { - "question": "Does this look correct?", - "answer": "yes", - "duration_ms": 30000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `question` | string | Question text | -| `answer` | string | Human's answer | -| `duration_ms` | number | Time waiting for answer | - -### `interview.timeout` - -Emitted when a human question times out. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "interview.timeout", - "node_id": "review", - "node_label": "review", - "properties": { - "question": "Does this look correct?", - "duration_ms": 300000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `question` | string | Question text | -| `duration_ms` | number | Time waited before timeout | - ---- - -## Checkpoint events - -### `checkpoint.completed` - -Emitted after a checkpoint is saved. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "checkpoint.completed", - "node_id": "code", - "node_label": "code", - "properties": { - "status": "succeeded", - "git_commit_sha": "abc123...", - "diff": "diff --git a/src/lib.rs b/src/lib.rs\n..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `status` | string | Checkpoint status | -| `git_commit_sha` | string? | Commit SHA at checkpoint time | -| `diff` | string? | Git diff captured for the checkpointed node | - -### `checkpoint.failed` - -Emitted when checkpoint saving fails. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "checkpoint.failed", - "node_id": "code", - "node_label": "code", - "properties": { - "error": "git commit failed: ..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `error` | string | Error message | - ---- - -## Git events - -### `git.commit` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "git.commit", - "node_id": "code", - "node_label": "code", - "properties": { - "sha": "abc123..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `sha` | string | Commit SHA | - -Note: `node_id` is optional — may be absent for non-stage commits. - -### `git.push` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "git.push", - "properties": { - "branch": "fabro/run-01JQXYZ", - "success": true - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `branch` | string | Branch name | -| `success` | boolean | Whether push succeeded | - -### `git.fetch` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "git.fetch", - "properties": { - "branch": "main", - "success": true - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `branch` | string | Branch name | -| `success` | boolean | Whether fetch succeeded | - -### `git.reset` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "git.reset", - "properties": { - "sha": "abc123..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `sha` | string | Target commit SHA | - ---- - -## Routing events - -### `edge.selected` - -Emitted when the engine selects the next edge to traverse. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "edge.selected", - "properties": { - "from_node": "code", - "to_node": "review", - "label": "tests_pass", - "condition": "outcome=succeeded", - "reason": "condition", - "preferred_label": "tests_pass", - "suggested_next_ids": ["review"], - "stage_status": "succeeded", - "is_jump": false - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `from_node` | string | Source node id | -| `to_node` | string | Target node id | -| `label` | string? | Edge label | -| `condition` | string? | Edge condition expression | -| `reason` | string | Selection reason (`"condition"`, `"preferred_label"`, `"jump"`, etc.) | -| `preferred_label` | string? | Stage's preferred label hint | -| `suggested_next_ids` | string[] | Stage's suggested next node ids | -| `stage_status` | string | Outcome status that influenced routing | -| `is_jump` | boolean | Whether this bypassed normal edge selection | - -### `loop.restart` - -Emitted when execution loops back to an earlier node. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "loop.restart", - "properties": { - "from_node": "review", - "to_node": "code" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `from_node` | string | Node that triggered the restart | -| `to_node` | string | Node to restart from | - ---- - -## Agent events - -Pebble's `CodingAgentEvent` stream is the agent event contract. Every event -the coding agent publishes for a stage, except streaming deltas, is stored -verbatim as an `EventBody::Agent` under a name derived from its variant -(`agent.message`, `agent.tool.started`, `agent.route.failover`, -`agent.mcp.server.ready`, `todo.created`, and so on; the full list is -`CODING_EVENT_NAMES`). Its `properties` are pebble's own envelope, so -pebble's event types are part of fabro's stored format, and the store folds -the same events into `StageProjection.agent` with pebble's -`SessionProjection`, the one fold of that stream. - -Fabro emits an agent event of its own only for a fact pebble cannot know: - -- `agent.session.activated` and `agent.session.deactivated`: the stage's - route, controls, permission level, and steering capabilities, as fabro - resolved them. -- `agent.tools.available`: the tool catalog fabro handed the agent. -- `agent.pair.user_message` and `agent.pair.system_message`: pair mode. -- `agent.interrupt.injected`, `agent.steer.buffered`, `agent.steer.dropped`: - run-level steering as it reaches, waits for, or misses a session. -- `agent.acp.started`, `agent.acp.completed`, `agent.acp.cancelled`, - `agent.acp.timed_out`: an external ACP agent process, which pebble does - not run. -- `prompt.failover`: a one-shot prompt stage moving to a fallback route, - which it does without pebble. - -Every agent activity event is stage-scoped and carries `node_id` (the -workflow stage), `node_label`, `stage_id`, `session_id`, and -`parent_session_id` in the envelope. Pebble's session lifecycle events -(`agent.session.started`, `agent.session.ended`) are stored with the stage -that ran the session like the rest. - -### `agent.session.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.session.started", - "session_id": "ses_abc", "parent_session_id": null, - "properties": { - "provider": "openai", - "model": "gpt-5.4" - } -} -``` - -Object-lifecycle event. `session_id` and `parent_session_id` are envelope fields. `properties.provider` and `properties.model` are optional. - -### `agent.session.activated` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.session.activated", - "node_id": "code", "node_label": "code", "stage_id": "code@1", - "session_id": "ses_abc", - "properties": { - "thread_id": "main", - "provider": "openai", - "model": "gpt-5.4", - "capabilities": ["steer"], - "visit": 1 - } -} -``` - -Stage-scoped lease event. A stage is steerable while the latest matching `agent.session.activated` lease is active. - -### `agent.session.deactivated` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.session.deactivated", - "node_id": "code", "node_label": "code", "stage_id": "code@1", - "session_id": "ses_abc", - "properties": { "visit": 1 } -} -``` - -Stage-scoped lease event. Consumers should pair it by `stage_id` and `session_id` so stale deactivations cannot clear a newer active lease. - -### `agent.session.ended` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.session.ended", - "session_id": "ses_abc", - "properties": {} -} -``` - -Object-lifecycle event. `session_id` and `parent_session_id` are envelope fields. No properties. - -### `agent.processing.end` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.processing.end", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": {} -} -``` - -No properties. One per prompt, when the agent has nothing more to do for -it. Pebble's `SessionProjection`, embedded in `StageProjection.agent`, reads -it to mark the prompt complete and the session idle, so the projection -rebuilt from the run's log needs it. Runs recorded before fabro stored it -never have it; their `agent.activity` stays `running`, and -`StageProjection.state` is the authority on whether the stage is done. - -### `agent.input` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.input", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "text": "Fix the login bug in auth.rs" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `text` | string | User input text | - -### `agent.llm.started` - -An inference request is about to be dispatched for this round. Emitted once -per round, after the request is built and compaction has run, immediately -before the stream is opened. - -`requested_model` is the canonical requested target, including an optional -speed tier. Failover can re-target mid-stage, so `agent.message` remains -authoritative for what actually answered. No usage or cost fields: neither -exists yet at this point. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.llm.started", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "requested_model": { - "provider": "anthropic", - "model_id": "claude-fable-5", - "speed": "fast" - }, - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `requested_model` | object | Requested provider, model ID, and optional speed tier | -| `visit` | number | Graph visit | - -### `agent.llm.first_output` - -The provider produced its first output for the current attempt. Edge-triggered -once per stream attempt; the latch re-arms when a broken or finish-less stream -replays the turn. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.llm.first_output", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "kind": "reasoning", - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `kind` | string | `reasoning`, `text`, or `tool_call` — observed, not inferred | -| `visit` | number | Graph visit | - -### `agent.message` - -Emitted when the assistant produces a complete message. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.message", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "text": "I've fixed the bug in auth.rs by...", - "model": "claude-sonnet-4-20250514", - "usage": { - "tokens": { - "input": 3000, - "output": 1500, - "reasoning": 200, - "cache_read": 1000, - "cache_write": 500 - }, - "cost": { "usd_micros": 12500, "source": "provider" } - }, - "tool_call_count": 2 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `text` | string | Assistant message text | -| `model` | string | Model identifier | -| `usage` | object | lithos-llm's `Usage` for this message, as pebble reported it | -| `usage.tokens` | object | The five disjoint token buckets: `input`, `output`, `reasoning`, `cache_read`, `cache_write` | -| `usage.cost` | object? | `usd_micros` and `source`, when the provider reported a cost | -| `usage.speed` | string? | Speed tier | -| `usage.raw` | object? | Raw provider-specific usage | -| `tool_call_count` | number | Number of tool calls in this turn | - -### `agent.tool.started` - -Emitted when the agent begins a tool call. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.tool.started", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "tool_name": "read_file", - "tool_call_id": "call_abc123", - "arguments": {"path": "src/auth.rs"} - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `tool_name` | string | Tool name | -| `tool_call_id` | string | Unique tool call id | -| `arguments` | object | Tool call arguments | - -### `agent.tool.completed` - -Emitted when a tool call finishes. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.tool.completed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "tool_name": "read_file", - "tool_call_id": "call_abc123", - "output": "fn login(user: &str) -> Result...", - "is_error": false - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `tool_name` | string | Tool name | -| `tool_call_id` | string | Unique tool call id | -| `output` | any | Tool output (string or structured) | -| `is_error` | boolean | Whether the tool returned an error | - -### `agent.tool.process.completed` - -Subordinate diagnostic for a tool call that ran a process, emitted between -`agent.tool.started` and `agent.tool.completed`. It explains the underlying -process outcome; `agent.tool.completed.is_error` remains the protocol and UI -truth. Absent when the tool never produced a process result (setup, transport, -or launch failure) and when the tool ran without a session-bound emitter. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.tool.process.completed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "tool_call_id": "call_abc123", - "properties": { - "exit_code": 7, - "termination": "exited", - "duration_ms": 812, - "streams_separated": true, - "exec_output_tail": {"stdout": "...", "stderr": "..."}, - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `exit_code` | integer | Process exit code; omitted for timeout and cancellation | -| `termination` | string | `exited`, `timed_out`, or `cancelled` | -| `duration_ms` | integer | Process duration | -| `streams_separated` | boolean | `false` when the provider could not separate stdout from stderr; the combined output is then in `exec_output_tail.stdout` | -| `exec_output_tail` | object | Bounded, redacted output tails; omitted when both streams were empty | -| `exec_output_tail.stdout` | string | Bounded stdout tail, or combined-output tail when `streams_separated` is `false`; omitted when empty | -| `exec_output_tail.stderr` | string | Bounded stderr tail; omitted when empty | -| `exec_output_tail.stdout_truncated` | boolean | `true` when earlier stdout bytes were omitted; omitted when `false` | -| `exec_output_tail.stderr_truncated` | boolean | `true` when earlier stderr bytes were omitted; omitted when `false` | -| `visit` | integer | Stage visit | - -### `agent.error` - -Emitted when the agent encounters an error. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.error", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "error": { ... } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `error` | object | AgentError (serialized) | - -### `agent.warning` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.warning", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "kind": "token_limit", - "message": "Approaching context window limit", - "details": {} - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `kind` | string | Warning kind | -| `message` | string | Warning message | -| `details` | object | Additional details | - -### `agent.loop.detected` - -Emitted when the agent detects a tool-use loop. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.loop.detected", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": {} -} -``` - -No properties. - -### `agent.steering.injected` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.steering.injected", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "text": "Remember to run tests after changes" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `text` | string | Injected steering text | - -### `agent.compaction.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.compaction.started", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "estimated_tokens": 50000, - "context_window_size": 128000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `estimated_tokens` | number | Estimated tokens before compaction | -| `context_window_size` | number | Model context window size | - -### `agent.compaction.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.compaction.completed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "original_turn_count": 40, - "preserved_turn_count": 10, - "summary_token_estimate": 2000, - "tracked_file_count": 5 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `original_turn_count` | number | Turns before compaction | -| `preserved_turn_count` | number | Turns preserved | -| `summary_token_estimate` | number | Token estimate for summary | -| `tracked_file_count` | number | Files being tracked | - -### `agent.llm.retry` - -Emitted when an attempt fails to open **or sustain** a stream and the turn is -replayed. The finish-less-stream case carries a synthetic `Stream` error and a -zero delay: the turn restarts even though no error was reported. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.llm.retry", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "provider": "anthropic", - "model": "claude-sonnet-4-20250514", - "attempt": 2, - "delay_secs": 1.5, - "phase": "open", - "error": { ... } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | LLM provider name | -| `model` | string | Model identifier | -| `attempt` | number | Retry attempt number, 0-based within the loop named by `phase` | -| `delay_secs` | number | Delay before retry in seconds | -| `phase` | string? | `open` (stream failed to open) or `consume` (stream broke or ended without a finish event). Absent on events stored before the discriminator existed | -| `error` | object | SdkError (serialized) | - -### `agent.sub.spawned` - -Emitted when a sub-agent is spawned. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.sub.spawned", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "agent_id": "sub_xyz", - "depth": 1, - "task": "Write unit tests for auth.rs" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `agent_id` | string | Sub-agent identifier | -| `depth` | number | Nesting depth | -| `task` | string | Task description | - -### `agent.sub.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.sub.completed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "agent_id": "sub_xyz", - "depth": 1, - "success": true, - "turns_used": 8 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `agent_id` | string | Sub-agent identifier | -| `depth` | number | Nesting depth | -| `success` | boolean | Whether the sub-agent succeeded | -| `turns_used` | number | Number of turns used | - -### `agent.sub.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.sub.failed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "agent_id": "sub_xyz", - "depth": 1, - "error": { ... } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `agent_id` | string | Sub-agent identifier | -| `depth` | number | Nesting depth | -| `error` | object | AgentError (serialized) | - -### `agent.sub.closed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.sub.closed", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "agent_id": "sub_xyz", - "depth": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `agent_id` | string | Sub-agent identifier | -| `depth` | number | Nesting depth | - -### `agent.memory.loaded` - -Emitted once per session right after memory discovery, before skills and MCP -initialization. The event is always emitted, even when no memory files are -loaded (in which case `files` is an empty array). Memory file **contents are -deliberately excluded** from the payload to keep the durable event stream free -of project documentation bytes; consumers that need contents must read the -files themselves. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.memory.loaded", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "provider_profile": "anthropic", - "files": [ - { - "path": "/repo/AGENTS.md", - "byte_count": 4096, - "loaded_bytes": 4096, - "truncated": false - } - ], - "total_loaded_bytes": 4096, - "budget_bytes": 32768, - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider_profile` | string | Active agent profile (`anthropic`, `openai`, `gemini`) | -| `files` | array | Discovered memory files. Empty when no memory was loaded. | -| `files[].path` | string | Absolute path of the memory file in the sandbox | -| `files[].byte_count` | number | Original file size in bytes | -| `files[].loaded_bytes` | number | Bytes actually loaded into the prompt budget | -| `files[].truncated` | boolean | `true` if the file was truncated to fit the budget | -| `total_loaded_bytes` | number | Sum of `files[].loaded_bytes` | -| `budget_bytes` | number | Total memory budget for the session (currently 32 KiB) | -| `visit` | number | Stage visit count | - -### `agent.skills.discovered` - -Emitted once per session right after skill discovery completes. The event is -always emitted, even when no skills are found (`skills` is an empty array). -Skills are sorted by name. `source_dirs` lists the directories that were -scanned in the configured precedence order. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.skills.discovered", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "provider_profile": "anthropic", - "source_dirs": [ - "/home/test/.fabro/skills", - "/repo/.fabro/skills", - "/repo/skills" - ], - "skills": [ - { "name": "commit", "description": "Make a commit" } - ], - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider_profile` | string | Active agent profile | -| `source_dirs` | array | Directories scanned for `SKILL.md` files (in precedence order) | -| `skills` | array | Discovered skills, sorted by `name`. Each entry is `{ name, description }`. | -| `visit` | number | Stage visit count | - -### `agent.skill.activated` - -Emitted whenever a skill is activated in the running session. Sources: - -- `slash` — the user input matched a `/skill-name` token and the skill template - was expanded inline. This event replaces the previous internal-only - `agent.skill.expanded` notification. -- `tool` — the model successfully called the `use_skill` tool and the skill - template was returned. Failed `use_skill` lookups (unknown names, missing - parameters) do **not** emit this event. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "agent.skill.activated", - "node_id": "code", "node_label": "code", - "session_id": "ses_abc", - "properties": { - "skill_name": "commit", - "source": "slash", - "visit": 1 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `skill_name` | string | Name of the activated skill | -| `source` | string | `"slash"` for `/skill-name` expansion, `"tool"` for `use_skill` activations | -| `visit` | number | Stage visit count | - -> `agent.skill.expanded` does not exist. The `AgentEvent::SkillExpanded` -> variant this note once described has since been removed from the code -> entirely; slash-skill expansion is reported through `agent.skill.activated` -> with `source == "slash"` instead. - -### `prompt.failover` - -Emitted by a one-shot prompt stage when it moves to a fallback route. The -prompt stage walks its fallback plan itself, so this is fabro's own event. -An agent stage never emits it: pebble walks the routes and reports each -move as `agent.route.failover`, stored verbatim (below). - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "prompt.failover", - "node_id": "summarize", - "node_label": "summarize", - "properties": { - "from_provider": "anthropic", - "from_model": "claude-sonnet-4-20250514", - "to_provider": "openai", - "to_model": "gpt-4o", - "attempt": 1, - "error": "rate limited" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `from_provider` | string | The provider that failed | -| `from_model` | string | The model that failed | -| `to_provider` | string | The provider the prompt continued on | -| `to_model` | string | The model the prompt continued on | -| `attempt` | number? | How many routes the prompt had moved through, this one included. Absent only on events recorded before it was kept | -| `error` | string | The failure that ended the previous route | - -Events recorded before this rename were named `agent.failover` and carried -`original_provider`, `original_model`, `requested_reasoning_effort`, -`effective_reasoning_effort`, and `continuation`; nothing read them. - -### `agent.route.failover`, `agent.mcp.server.ready`, `agent.mcp.server.failed`, `agent.mcp.server.disconnected` - -Pebble's `RouteFailover`, `McpServerReady`, `McpServerFailed`, and -`McpServerDisconnected` events, stored verbatim with pebble's envelope in -`properties` like every other pebble event. They are the only record of an -agent stage's route moves and MCP server outcomes: the stage view reads -both from `StageProjection.agent`, which they feed. Runs recorded before -fabro stored them carry fabro's former mirrors, `agent.failover`, -`agent.mcp.ready`, `agent.mcp.failed`, and `agent.mcp.disconnected`, -which no reader folds any more. - -### `agent.route.failover.stopped` - -Pebble's `RouteFailoverStopped` event, stored verbatim like every other -pebble event. An agent stage with fallback routes -publishes it when a model failure ends the prompt on its current route -anyway: the failure does not qualify for failover (`reason: "ineligible"`) -or every route has been taken (`reason: "exhausted"`). It follows the -`agent.error` that reports the failure; a stage without fallback routes and -a cancelled prompt publish nothing here. The properties are pebble's -envelope (`seq`, `stream_id`, `session_id`, `timestamp`) plus -`event.RouteFailoverStopped` with `route`, `attempt`, `reason`, and `error`. - -### Agent events that are never serialized - -`AgentEvent` also has variants that exist only on the agent session's -in-process broadcast channel. `is_streaming_noise()` filters them out before -the workflow emitter builds a `RunEvent`, so they never reach the run store, -SSE, `fabro events`, or a JSONL sink — they have no envelope, and no external -consumer can observe them: - -- `AssistantOutputReplace` — clears in-progress output buffers when a turn is - replayed -- `TextDelta`, `ReasoningDelta` — streaming assistant chunks -- `ToolCallOutputDelta` — streaming tool output chunks - -They were previously documented here as though they were durable events, with -full envelope examples. If any of them ever needs to be durable, it belongs in -a separate transient stream rather than the canonical persisted contract — -long autonomous runs would generate orders of magnitude more delta traffic -than the interactive sessions surface handles. - ---- - -## Subgraph events - -### `subgraph.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "subgraph.started", - "node_id": "pipeline", - "node_label": "pipeline", - "properties": { - "start_node": "sub_start" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `start_node` | string | First node in the subgraph | - -### `subgraph.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "subgraph.completed", - "node_id": "pipeline", - "node_label": "pipeline", - "properties": { - "steps_executed": 4, - "status": "succeeded", - "duration_ms": 25000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `steps_executed` | number | Number of steps executed | -| `status` | string | Subgraph outcome status | -| `duration_ms` | number | Subgraph duration | - ---- - -## Sandbox events - -Sandbox events have the nested `SandboxEvent` unwrapped into `properties`. - -### `sandbox.initializing` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.initializing", - "properties": { - "provider": "daytona" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | - -### `sandbox.ready` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.ready", - "properties": { - "provider": "daytona", - "duration_ms": 5000, - "name": "sandbox-01JQXYZ", - "cpu": 4.0, - "memory": 8.0, - "url": "https://sandbox.example.com" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | -| `duration_ms` | number | Initialization duration | -| `name` | string? | Sandbox instance name | -| `cpu` | number? | CPU cores allocated | -| `memory` | number? | Memory in GB allocated | -| `url` | string? | Sandbox URL | - -### `sandbox.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.failed", - "properties": { - "provider": "daytona", - "error": "workspace creation failed", - "duration_ms": 3000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | -| `error` | string | Error message | -| `duration_ms` | number | Time before failure | - -### `sandbox.initialized` - -Emitted after the engine completes sandbox initialization (distinct from `sandbox.ready` which comes from the sandbox provider). - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.initialized", - "properties": { - "working_directory": "/workspace/my-project", - "provider": "daytona", - "identifier": "sandbox-123", - "repo_cloned": true, - "clone_origin_url": "https://github.com/acme/my-project.git", - "clone_branch": "main" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `working_directory` | string | Working directory inside sandbox | -| `provider` | string | Sandbox provider | -| `identifier` | string? | Provider-specific sandbox identifier | -| `repo_cloned` | boolean? | Whether the provider cloned a repository into the sandbox | -| `clone_origin_url` | string? | Repository URL cloned into the sandbox, with credentials removed | -| `clone_branch` | string? | Branch requested for the sandbox clone | - -### `sandbox.cleanup.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.cleanup.started", - "properties": { - "provider": "daytona" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | - -### `sandbox.cleanup.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.cleanup.completed", - "properties": { - "provider": "daytona", - "duration_ms": 2000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | -| `duration_ms` | number | Cleanup duration | - -### `sandbox.cleanup.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.cleanup.failed", - "properties": { - "provider": "daytona", - "error": "workspace not found" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `provider` | string | Sandbox provider name | -| `error` | string | Error message | - -### Sandbox driver events - -Everything the sandbox driver reports about a run's sandbox is stored whole. The -event name derives from the driver's event: `..` for an -operation (`sandbox.start.started`, `sandbox.stop.completed`, `sandbox.delete.failed`, -`sandbox.create.progress` for an image pull inside the create, `snapshot.create.started` -and `snapshot.create.completed` for a snapshot build), `.state` for a state -observation, and `.notice` for a notice. `properties` is the driver's event as -the driver serializes it. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.stop.completed", - "properties": { - "id": {"source_id": "9b2f…", "sequence": 4}, - "occurred_at": "2026-08-31T20:00:00Z", - "provider": "docker", - "subject": {"type": "sandbox", "id": "container-abc123"}, - "operation_id": "58a1…", - "correlation_id": "01JQ…", - "type": "operation_completed", - "action": "stop", - "duration": {"secs": 1, "nanos": 250000000} - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `id` | object | The driver's event id: `source_id` and `sequence` within that source | -| `occurred_at` | string | When the driver observed the event (RFC 3339) | -| `provider` | string | The driver's provider kind (`host`, `docker`, `daytona`, a plugin's kind) | -| `subject` | object | `type` (`sandbox`, `snapshot`, `volume`, `provider`) with the resource's `id` and `name` when known | -| `operation_id` | string | Groups the started, progress, and completed or failed events of one operation | -| `correlation_id` | string | The run id fabro attached | -| `type` | string | `operation_started`, `operation_progress`, `operation_completed`, `operation_failed`, `state_observed`, or `notice` | -| `action` | string | The operation (`create`, `start`, `stop`, `delete`, `snapshot`, …) on operation events | -| `progress` | object | `code` (`image.pull`, `snapshot.build`, …), `message`, and optional `completed`, `total`, `unit` on progress events | -| `duration` | object | `secs` and `nanos` on completed and failed events | -| `error` | object | `kind`, `message`, `retryable`, `causes` on failed events | - -Events stored under `sandbox.start.*`, `sandbox.stop.*`, `sandbox.delete.*`, and -`sandbox.snapshot.*` before the driver's events were kept whole carry fabro's earlier -`provider`, `name`, `duration_ms`, and `error` properties instead; readers treat them as -unknown bodies. - -### `sandbox.git.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.git.started", - "properties": { - "url": "https://github.com/org/repo.git", - "branch": "main" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `url` | string | Repository URL | -| `branch` | string? | Branch to clone | - -### `sandbox.git.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.git.completed", - "properties": { - "url": "https://github.com/org/repo.git", - "duration_ms": 8000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `url` | string | Repository URL | -| `duration_ms` | number | Clone duration | - -### `sandbox.git.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "sandbox.git.failed", - "properties": { - "url": "https://github.com/org/repo.git", - "error": "authentication failed" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `url` | string | Repository URL | -| `error` | string | Error message | - ---- - -## Setup events - -### `setup.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "setup.started", - "properties": { - "command_count": 3 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `command_count` | number | Number of setup commands | - -### `setup.command.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "setup.command.started", - "properties": { - "command": "npm install", - "index": 0 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `command` | string | Command being run | -| `index` | number | Command index | - -### `setup.command.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "setup.command.completed", - "properties": { - "command": "npm install", - "index": 0, - "exit_code": 0, - "duration_ms": 5000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `command` | string | Command that ran | -| `index` | number | Command index | -| `exit_code` | number | Process exit code | -| `duration_ms` | number | Command duration | - -### `setup.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "setup.completed", - "properties": { - "duration_ms": 15000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `duration_ms` | number | Total setup duration | - -### `setup.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "setup.failed", - "properties": { - "command": "npm install", - "index": 1, - "exit_code": 1, - "stderr": "npm ERR! ..." - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `command` | string | Command that failed | -| `index` | number | Command index | -| `exit_code` | number | Process exit code | -| `stderr` | string | Standard error output | - ---- - -## CLI ensure events - -These legacy events may appear in older run logs. Current CLI backend runs do not emit them because Fabro no longer installs or prepares provider CLIs at stage runtime. - -### `cli.ensure.started` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "cli.ensure.started", - "properties": { - "cli_name": "aider", - "provider": "openai" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `cli_name` | string | CLI tool name | -| `provider` | string | LLM provider | - -### `cli.ensure.completed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "cli.ensure.completed", - "properties": { - "cli_name": "aider", - "provider": "openai", - "already_installed": true, - "node_installed": false, - "duration_ms": 500 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `cli_name` | string | CLI tool name | -| `provider` | string | LLM provider | -| `already_installed` | boolean | Whether it was already present | -| `node_installed` | boolean | Whether Node.js was installed | -| `duration_ms` | number | Duration | - -### `cli.ensure.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "cli.ensure.failed", - "properties": { - "cli_name": "aider", - "provider": "openai", - "error": "pip install failed", - "duration_ms": 3000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `cli_name` | string | CLI tool name | -| `provider` | string | LLM provider | -| `error` | string | Error message | -| `duration_ms` | number | Duration | - ---- - -## Pull request events - -### `pull_request.creation_requested` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "pull_request.creation_requested", - "properties": { - "creation_id": "01KYYK70WTZT2E551P3H5P0059", - "model": "gpt-5.4", - "force": false - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `creation_id` | string | Stable identifier for this pull request creation request | -| `model` | string | Resolved model identifier used to generate the pull request content | -| `force` | boolean | Whether creation is allowed for a run without a successful conclusion | - -### `pull_request.created` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "pull_request.created", - "properties": { - "pr_url": "https://github.com/org/repo/pull/42", - "pr_number": 42, - "head_sha": "d34db33f", - "draft": true - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `pr_url` | string | Pull request URL | -| `pr_number` | number | Pull request number | -| `head_sha` | string (optional) | Verified commit SHA at the remote PR head; absent on older events | -| `draft` | boolean | Whether the PR is a draft | - -### `pull_request.linked` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "pull_request.linked", - "properties": { - "pull_request": { - "provider": "github", - "html_url": "https://github.com/org/repo/pull/42", - "number": 42, - "owner": "org", - "repo": "repo", - "title": "Review deployment chart" - } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `pull_request` | object | Stored GitHub pull request association. `title`, `base_branch`, and `head_branch` may be included when live GitHub metadata is available. | - -### `pull_request.unlinked` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "pull_request.unlinked", - "properties": { - "pull_request": { - "provider": "github", - "html_url": "https://github.com/org/repo/pull/42", - "number": 42 - } - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `pull_request` | object | Pull request association removed from the run. | - -### `pull_request.failed` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "pull_request.failed", - "properties": { - "creation_id": "01KYYK70WTZT2E551P3H5P0059", - "error": "insufficient permissions" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `creation_id` | string (optional) | Explicit pull request creation this failure resolves. Absent for publish-stage failures. | -| `error` | string | Error message | - -When `creation_id` names the run's pending pull request creation, the run -projection marks that creation `failed`. A `pull_request.failed` event without -a `creation_id` (the workflow publish stage) does not change creation state. - -## Artifact events - -### `artifact.captured` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "artifact.captured", - "node_id": "code", - "node_label": "code", - "properties": { - "attempt": 1, - "node_slug": "code", - "path": "screenshot.png", - "mime": "image/png", - "content_md5": "d41d8cd98f00b204e9800998ecf8427e", - "content_sha256": "e3b0c44298fc1c149afbf4c8996fb924...", - "bytes": 45000 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `attempt` | number | Attempt number | -| `node_slug` | string | Node slug for asset path | -| `path` | string | Asset file path | -| `mime` | string | MIME type | -| `content_md5` | string | MD5 hash | -| `content_sha256` | string | SHA-256 hash | -| `bytes` | number | File size in bytes | - ---- - -## SSH events - -### `ssh.ready` - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "ssh.ready", - "properties": { - "ssh_command": "ssh user@host -p 2222" - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `ssh_command` | string | SSH command to connect | - ---- - -## Watchdog events - -### `watchdog.timeout` - -Emitted when the stall watchdog detects no progress. - -```json -{ - "id": "...", "ts": "...", "run_id": "...", - "event": "watchdog.timeout", - "node_id": "code", - "node_label": "code", - "properties": { - "idle_seconds": 1800 - } -} -``` - -| Property | Type | Description | -|----------|------|-------------| -| `idle_seconds` | number | Seconds since last activity | diff --git a/docs/internal/fabro-event-schema-v2-concrete-shape.md b/docs/internal/fabro-event-schema-v2-concrete-shape.md deleted file mode 100644 index 89b8ee806..000000000 --- a/docs/internal/fabro-event-schema-v2-concrete-shape.md +++ /dev/null @@ -1,488 +0,0 @@ -# Fabro Event Schema V2: Concrete Shape - -Date: 2026-04-09 - -Status: implemented - -This document turns the settled design decisions from the event-schema discussion into a concrete wire-contract proposal. - -It intentionally supersedes the earlier framing in [fabro-event-schema-v2-proposal.md](/Users/bhelmkamp/p/fabro-sh/fabro/docs-internal/fabro-event-schema-v2-proposal.md) for: - -- proposal 1: one canonical persisted log, not two truths -- proposal 2: formalize and generalize the existing `since_seq` replay contract, rather than inventing replay from scratch - -## Design Decisions Carried Forward - -- one canonical persisted event log -- plain hand-coded Rust structs are the authoritative source of truth for the event contract -- `RunEvent` remains the canonical semantic event type -- `seq` remains outside `RunEvent`, in the store/API envelope -- replay stays built around ordered `since_seq` cursors -- typed Rust consumers matching on `EventBody` remain the primary consumer model -- the envelope widens only modestly for execution topology and tool-call correlation: `stage_id`, `parallel_group_id`, `parallel_branch_id`, `tool_call_id` -- existing durable event families stay broadly intact -- live token/delta noise does not become part of the durable persisted Rust event contract -- snapshots are out of scope for both the durable event contract and the attach API - -## Contract Source Of Truth - -V2 does not adopt schema generation or a registry-first workflow. - -The authoritative source of truth for the event contract should be plain, hand-coded Rust structs and enums that model the public wire shape directly. - -Implications: - -- the Rust event types are the canonical contract -- this document describes that contract and should stay aligned with the Rust types -- any TypeScript types, JSON Schema, or OpenAPI fragments are secondary artifacts, not the source of truth -- codegen is explicitly out of scope for the initial V2 implementation - -## Why Evolve The Current Model - -V2 should evolve Fabro's existing event architecture rather than replace it with a generic event platform. - -Earlier drafts of this document proposed a generic reducer contract, a larger ontology-first envelope, and a narrower replacement event catalog. V2 walks that back. The current code's boundary between internal workflow events, `RunEvent`, and `EventEnvelope` is stronger and simpler than it first appeared, so evolving that model is cheaper and clearer than replacing it. - -The current code already has a strong separation of concerns: - -- internal workflow/runtime events in `fabro-workflow` -- one canonical semantic `RunEvent` -- a store/API envelope that carries `seq` outside the event payload - -That separation is worth preserving. The main V2 changes should be: - -- modest envelope widening for execution topology -- cleanup and clarification of event-family boundaries -- keeping the durable event catalog semantic and typed - -V2 should not introduce: - -- a generic reducer contract based on `entity_type` / `event_role` -- canonical persisted token deltas -- snapshot events as a second truth layer - -## Capability Coverage Decisions - -V2 is evolutionary over the current `RunEvent` surface. It keeps the existing durable event families broadly intact rather than replacing them with a new ontology. - -The main additions are: - -- `stage_id` in the envelope for concrete stage execution identity -- `parallel_group_id` in the envelope for one execution of a parallel node -- `parallel_branch_id` in the envelope for one branch inside a parallel execution -- `tool_call_id` in the envelope for agent tool lifecycle events that need a stable cross-family join key - -Everything else should remain in typed `EventBody` props unless there is a strong cross-family reason to promote it. `session_id` already exists in the envelope today and stays as-is. `tool_call_id` is promoted now because `agent.tool.*` events already carry a stable tool-call identity that other durable families can reference when needed. `turn_id` is deferred because Fabro does not yet have a durable turn identity that spans the families that would need to join on it. - -## Exact Delta From Current Code - -This is the implementation delta from the current Rust codebase, not the full history of how the design was reached. - -### Add - -- add `stage_id: Option` to `RunEvent` -- add `parallel_group_id: Option` to `RunEvent` -- add `parallel_branch_id: Option` to `RunEvent` -- add `tool_call_id: Option` to `RunEvent` -- add `actor: Option` to `RunEvent` -- extend envelope extraction in `stored_event_fields()` to populate the new execution-topology fields when known -- extend envelope extraction in `stored_event_fields()` to populate `tool_call_id` on tool-lifecycle events when known -- update `RunEvent` serialization and parsing so the new optional envelope fields round-trip cleanly - -### Keep As-Is - -- `RunEvent` remains the canonical semantic event type -- `EventBody` remains the typed tagged union of durable event families -- `EventBody::Unknown` remains the compatibility valve for unknown event names on read -- `EventEnvelope` remains the ordered outer wrapper with `seq` outside the event payload -- `EventEnvelope.payload` remains `EventPayload`, not `RunEvent` -- the internal/store `EventEnvelope` Rust type stays wrapped as `{ seq, payload }` -- attach/replay remains exact ordered replay from `since_seq`, followed by live tailing -- current durable event families stay broadly intact -- live token/delta noise remains outside the durable persisted contract -- snapshots remain out of scope - -### Do Not Do - -- do not inline `seq` into `RunEvent` -- do not introduce `entity_type`, `entity_id`, or `event_role` -- do not replace typed Rust consumers with a generic reducer model -- do not redesign the store envelope -- do not add snapshot events or attach-time synthetic snapshots -- do not persist token deltas or other live UI noise as durable `RunEvent`s - -## Canonical Rust Shapes - -V2 should model the public contract directly as hand-coded Rust types, following the existing architecture. - -```rust -pub struct RunEvent { - pub id: String, - pub ts: DateTime, - pub run_id: RunId, - pub node_id: Option, - pub node_label: Option, - pub stage_id: Option, - pub parallel_group_id: Option, - pub parallel_branch_id: Option, - pub session_id: Option, - pub parent_session_id: Option, - pub tool_call_id: Option, - pub actor: Option, - pub body: EventBody, -} - -pub struct EventEnvelope { - pub seq: u32, - pub payload: EventPayload, -} - -pub struct ActorRef { - pub kind: ActorKind, - pub id: Option, - pub display: Option, -} - -pub enum ActorKind { - User, - Agent, - System, -} -``` - -`RunEvent` remains the semantic product event. `EventEnvelope` remains the ordered store/API wrapper. The store continues to persist validated JSON `EventPayload`, not typed `RunEvent` structs. - -For wire JSON, `EventEnvelope` should serialize in flattened form so clients see: - -```json -{ - "seq": 4861, - "id": "...", - "ts": "...", - "run_id": "...", - "event": "...", - "properties": { ... } -} -``` - -That flattening is a wire concern only. It does not move `seq` into `RunEvent`, and it does not change the internal/store Rust shape of `EventEnvelope`. - -`EventBody` remains a hand-coded tagged enum serialized as: - -```json -{ - "event": "stage.completed", - "properties": { "...": "..." } -} -``` - -V2 should also preserve the current unknown-event fallback shape: - -```rust -EventBody::Unknown { - name: String, - properties: serde_json::Value, -} -``` - -This fallback already exists in the current code and should be kept. - -### Envelope Rules - -- `id`, `ts`, `run_id`, and `event` are always present on the serialized `RunEvent`. -- `seq` is not part of `RunEvent`. It stays in the outer `EventEnvelope`. -- Optional envelope fields are omitted, never serialized as `null`. -- The existing top-level envelope fields remain: - - `node_id` - - `node_label` - - `session_id` - - `parent_session_id` -- V2 adds only these new optional envelope fields: - - `stage_id` - - `parallel_group_id` - - `parallel_branch_id` - - `tool_call_id` -- Other relationship identifiers stay inside typed `properties`. -- `turn_id` remains in typed `properties`; see the deferral decision in `Capability Coverage Decisions`. -- `actor` is optional. When present, it identifies the primary actor for the event. -- Set `actor` on human- or agent-initiated events where that identity matters to consumers. Example: `run.cancel.requested` should identify the user who initiated the cancel. -- Set `actor` on durable agent output when the producing session identity matters. Example: `agent.message` should identify the agent session. -- Omit `actor` for routine runtime events with no meaningful primary actor. Example: `stage.started`. - -### ID Format Conventions - -- `run_id` keeps Fabro's current format: an unprefixed ULID string. -- `stage_id` keeps Fabro's current format: `"{node_id}@{visit}"`. -- `node_id` is the stable graph node identifier from the workflow definition. -- `parallel_group_id` should be the durable identity of one execution of a parallel node. The default format should be `"{node_id}@{visit}"`. -- `parallel_branch_id` should be the durable identity of one branch within a parallel execution. The default format should be `"{parallel_group_id}:{index}"`. -- Consumers should otherwise treat IDs as opaque strings. - -### Presence Expectations - -- `stage_id` is present on events tied to a concrete stage execution. -- `parallel_group_id` is present on `parallel.*` events and on events emitted inside a parallel execution when that scope is known. -- `parallel_branch_id` is present on `parallel.branch.*` events and on nested events emitted inside a specific branch when that scope is known. -- `session_id` and `parent_session_id` keep their current meaning for forwarded agent/session activity. -- `tool_call_id` is present on `agent.tool.*` events and on other durable events that directly describe the same tool call. -- `node_label` remains in the envelope for display-oriented consumers. -- `actor` is expected on control actions and durable agent output when there is a meaningful user or agent identity to expose. It is usually omitted on routine runtime lifecycle events. - -## Consumer Model - -Rust consumers should keep matching on `RunEvent.body` using typed `EventBody` variants. - -This document does not adopt: - -- `entity_type` -- `entity_id` -- `event_role` -- a generic reducer contract - -External JSON consumers should continue to: - -- match on `"event"` -- read event-specific values from `"properties"` -- read `"seq"` from the flattened outer event envelope on API/SSE responses -- use envelope metadata only for cross-cutting context such as stage, session, execution topology, and tool-call correlation - -## Replay Contract - -Fabro keeps the current replay model: - -- ordered events are stored as `EventEnvelope { seq, payload }` -- API/SSE serialization of `EventEnvelope` should flatten `seq` into the top-level JSON object returned to clients -- attach starts from `since_seq` -- the server replays exact persisted envelopes and then tails live envelopes while the run is active -- SSE keepalive comments are transport frames, not events - -V2 does not introduce: - -- `run.snapshot` -- `session.snapshot` -- API-level attach snapshots -- persisted snapshot events of any kind - -The durable model remains simple: replay ordered events, no duplicate truth layer. - -## Implementation Checklist - -An engineer implementing this proposal should make only these structural changes unless a later section explicitly says otherwise. - -1. Update [`RunEvent`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/mod.rs) to add: - - `stage_id` - - `parallel_group_id` - - `parallel_branch_id` - - `tool_call_id` - - `actor` -2. Update `RunEvent::to_value()` and `RunEvent` parsing in [`run_event/mod.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/mod.rs) so the new envelope fields serialize and deserialize. -3. Extend `StoredEventFields` and `stored_event_fields()` in [`event.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/components/fabro-workflow/src/event.rs) to populate: - - `stage_id` - - `parallel_group_id` - - `parallel_branch_id` - - `tool_call_id` on tool-lifecycle events - - `actor` when there is a clear primary actor - These values should come from the emitter's current execution context for stage and parallel scope, and from event-specific payloads for `tool_call_id`. -4. Leave [`EventEnvelope`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/components/fabro-store/src/types.rs) structurally unchanged: - - `seq: u32` - - `payload: EventPayload` -5. Update API/SSE envelope serialization so wire JSON is flattened: - - top-level `seq` - - then the `RunEvent` payload fields alongside it - - no `"payload": { ... }` wrapper in JSON responses -6. Leave the replay/attach flow unchanged in behavior: - - persisted replay from `since_seq` - - live tail after replay - - no snapshots -7. Keep the current `EventBody` family surface unless there is an explicit product reason to change a specific family. -8. Keep streaming-noise agent events out of durable `RunEvent` conversion. -9. Update the HTTP/API schema docs to reflect both: - - new `RunEvent` envelope fields - - flattened JSON serialization of `EventEnvelope` - -## EventBody And Property Model - -V2 should keep the current hand-coded domain split for prop structs: - -- run props in [`run.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/run.rs) -- stage and checkpoint props in [`stage.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/stage.rs) -- agent props in [`agent.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/agent.rs) -- infra/setup props in [`infra.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/infra.rs) -- parallel/interview/git/misc props in [`misc.rs`](/Users/bhelmkamp/p/fabro-sh/fabro/lib/foundation/fabro-types/src/run_event/misc.rs) - -That split is part of the design quality. V2 should keep adding hand-coded prop structs, not collapse everything into generic maps. - -## Durable Event Surface - -V2 keeps the current durable family surface broadly intact. - -### Run - -- `run.created` -- `run.started` -- `run.submitted` -- `run.starting` -- `run.running` -- `run.removing` -- `run.cancel.requested` -- `run.pause.requested` -- `run.unpause.requested` -- `run.paused` -- `run.unpaused` -- `run.rewound` -- `run.completed` -- `run.failed` -- `run.notice` - -### Stage And Prompt - -- `stage.started` -- `stage.completed` -- `stage.failed` -- `stage.retrying` -- `stage.prompt` -- `prompt.completed` - -### Parallel - -- `parallel.started` -- `parallel.branch.started` -- `parallel.branch.completed` -- `parallel.completed` - -### Interview / Human Input - -- `interview.started` -- `interview.completed` -- `interview.timeout` -- `interview.interrupted` - -### Checkpoint - -- `checkpoint.completed` -- `checkpoint.failed` - -### Agent Durable Events - -Pebble's events, stored verbatim under the names `fabro_types::CODING_EVENT_NAMES` -lists (every `CodingEvent` variant except the streaming deltas): - -- `agent.session.started`, `agent.session.ended`, `agent.processing.end` -- `agent.input`, `agent.message` -- `agent.llm.started`, `agent.llm.first_output`, `agent.llm.retry` -- `agent.tool.started`, `agent.tool.completed`, `agent.tool.process.completed`, `agent.tool.rounds.exhausted` -- `agent.error`, `agent.warning`, `agent.loop.detected` -- `agent.steering.injected`, `agent.round.interrupted` -- `agent.compaction.started`, `agent.compaction.completed`, `agent.compaction.failed`, `agent.compaction.cancelled` -- `agent.route.failover`, `agent.route.failover.stopped` -- `agent.mcp.server.ready`, `agent.mcp.server.failed`, `agent.mcp.server.disconnected` -- `agent.sub.spawned`, `agent.sub.turn.started`, `agent.sub.completed`, `agent.sub.failed`, `agent.sub.closed` -- `agent.memory.loaded`, `agent.skills.discovered`, `agent.skill.activated` -- `todo.created`, `todo.updated`, `todo.deleted` - -Fabro's own, for facts pebble cannot know: - -- `agent.session.activated`, `agent.session.deactivated`, `agent.tools.available` -- `agent.pair.user_message`, `agent.pair.system_message` -- `agent.interrupt.injected`, `agent.steer.buffered`, `agent.steer.dropped` -- `agent.acp.started`, `agent.acp.completed`, `agent.acp.cancelled`, `agent.acp.timed_out` -- `prompt.failover` (a one-shot prompt stage's move to a fallback route) - -The former mirrors `agent.mcp.ready`, `agent.mcp.failed`, -`agent.mcp.disconnected`, and `agent.failover` are no longer emitted; runs -recorded with them read them back as generic events. - -### Git - -- `git.commit` -- `git.push` -- `git.branch` -- `git.worktree.added` -- `git.worktree.removed` -- `git.fetch` -- `git.reset` - -### Infra And Execution - -- `sandbox.*` -- `setup.*` -- `cli.ensure.*` (legacy only) -- `command.*` -- `agent.cli.*` -- `pull_request.*` -- `artifact.captured` -- `ssh.ready` -- `subgraph.*` -- `edge.selected` -- `loop.restart` -- `retro.*` - -## Explicitly Non-Durable Streaming Noise - -The current boundary that keeps live token/delta noise out of `RunEvent` should remain in place. - -These stay outside the durable persisted contract: - -- `agent.output.replace` -- `agent.text.delta` -- `agent.reasoning.delta` -- `agent.tool.output.delta` - -(`agent.skill.expanded` was previously listed here. No such event exists — the -`AgentEvent::SkillExpanded` variant was removed, and slash-skill expansion is -reported through the durable `agent.skill.activated` with `source == "slash"`.) - -If Fabro needs those for UI, they belong in a separate transient stream, not in the canonical persisted Rust event contract. - -## Example Shapes - -### Flattened Wire JSON - -```json -{ - "seq": 4861, - "id": "evt_01JSE1N7RJD1NW2JSDT3W0YQ92", - "ts": "2026-04-08T16:21:11.106Z", - "run_id": "01JSE1M0Q0P8P6KQW9Q6D58Q0E", - "event": "agent.tool.completed", - "stage_id": "code@1", - "node_id": "code", - "node_label": "Code", - "session_id": "ses_child", - "tool_call_id": "call_1", - "parent_session_id": "ses_parent", - "properties": { - "tool_name": "read_file", - "output": { - "summary": "Read docs-internal/events-strategy.md" - }, - "is_error": false, - "visit": 1 - } -} -``` - -In Rust, `EventEnvelope` still remains `{ seq, payload: EventPayload }`. The example above is only the flattened API/SSE JSON form of that envelope. - -## Practical Guidance - -- Preserve the current one-time canonicalization boundary from internal `Event` to external `RunEvent`. -- Keep `RunEvent` semantic and typed. Do not turn it into a generic reducer envelope. -- Keep `seq` outside the event payload. -- Widen the envelope only modestly: `stage_id`, `parallel_group_id`, `parallel_branch_id`, and `tool_call_id`. -- Keep `session_id` as the existing top-level session field. -- Keep event-specific detail inside typed props. -- Preserve `EventBody::Unknown` as the compatibility valve for unknown event names on read. -- Do not store token deltas or other live UI noise as durable `RunEvent`s. -- Do not add snapshot events or attach-time synthetic snapshots. -- When adding a new durable event, update the current Rust boundary cleanly: - - internal `Event` - - `event_name()` - - envelope extraction - - `EventBody` - - typed props - - affected consumers - -## Open Follow-Up - -- `correlation_id`-style cross-entity grouping remains deferred until Fabro has a concrete consumer and explicit propagation rules diff --git a/lib/apps/fabro-cli/tests/it/cmd/runner.rs b/lib/apps/fabro-cli/tests/it/cmd/runner.rs index 46a32f3a5..e05d39fbe 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/runner.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/runner.rs @@ -19,8 +19,8 @@ use fabro_types::{FailureReason, RunStreamItem, StageId}; use httpmock::MockServer; use super::support::{ - command_log_text, created_run_id, find_run_dir, local_dev_token, output_stderr, run_events, - run_state, server_endpoint, server_target, wait_for_lifecycle, wait_for_status, + command_log_text, created_run_id, find_run_dir, local_dev_token, output_stderr, run_state, + run_stream_items, server_endpoint, server_target, wait_for_lifecycle, wait_for_status, write_gated_workflow, }; use crate::support::{issue_test_worker_jwt, seed_dev_token_auth, unique_run_id}; @@ -36,7 +36,7 @@ fn auth_context() -> fabro_test::TestContext { } fn stored_worker_events(run_dir: &std::path::Path) -> Vec { - run_events(run_dir) + run_stream_items(run_dir) } /// The platform record of `item`, when it carries one. diff --git a/lib/apps/fabro-cli/tests/it/cmd/support.rs b/lib/apps/fabro-cli/tests/it/cmd/support.rs index 8333bf934..fb4d73002 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/support.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/support.rs @@ -472,7 +472,7 @@ pub(crate) fn setup_detached_dry_run(context: &TestContext) -> RunSetup { let run_id = created_run_id(&output); let run = resolve_run(context, &run_id); let deadline = Instant::now() + command_timeout(); - while run_events(&run.run_dir).is_empty() { + while run_stream_items(&run.run_dir).is_empty() { assert!( Instant::now() < deadline, "timed out waiting for store events for {run_id}" @@ -852,7 +852,7 @@ pub(crate) fn run_state(run_dir: &Path) -> RunProjection { )) } -pub(crate) fn run_events(run_dir: &Path) -> Vec { +pub(crate) fn run_stream_items(run_dir: &Path) -> Vec { let run_id = infer_run_id(run_dir); let response: serde_json::Value = block_on(get_server_json( run_dir, @@ -907,7 +907,7 @@ pub(crate) fn wait_for_lifecycle(run_dir: &Path, transition: &str) { fn wait_for_stream_item(run_dir: &Path, what: &str, matches: impl Fn(&RunStreamItem) -> bool) { let deadline = std::time::Instant::now() + command_timeout(); loop { - if run_events(run_dir).iter().any(&matches) { + if run_stream_items(run_dir).iter().any(&matches) { return; } assert!( diff --git a/lib/apps/fabro-cli/tests/it/cmd/worker_auth.rs b/lib/apps/fabro-cli/tests/it/cmd/worker_auth.rs index 1d70e45f5..ec2d99b9e 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/worker_auth.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/worker_auth.rs @@ -249,7 +249,11 @@ async fn wait_for_http_ready(base_url: &str, child: &mut Child) { } } -async fn run_events(api_base_url: &str, run_id: &str, access_token: &str) -> Vec { +async fn run_stream_items( + api_base_url: &str, + run_id: &str, + access_token: &str, +) -> Vec { let response = fabro_test::test_http_client() .get(format!( "{api_base_url}/api/v1/runs/{run_id}/events?after=0&limit=1000" @@ -274,7 +278,7 @@ async fn wait_for_completed_events( ) -> Vec { let deadline = Instant::now() + COMMAND_TIMEOUT; loop { - let events = run_events(api_base_url, run_id, access_token).await; + let events = run_stream_items(api_base_url, run_id, access_token).await; if events.iter().any(is_terminal_lifecycle) { return events; } diff --git a/lib/apps/fabro-cli/tests/it/workflow/mod.rs b/lib/apps/fabro-cli/tests/it/workflow/mod.rs index b6e3a8a37..f9cd01812 100644 --- a/lib/apps/fabro-cli/tests/it/workflow/mod.rs +++ b/lib/apps/fabro-cli/tests/it/workflow/mod.rs @@ -56,7 +56,7 @@ pub(super) fn completed_nodes(run_dir: &Path) -> Vec { } pub(super) fn has_event(run_dir: &Path, event_name: &str) -> bool { - run_events(run_dir) + run_stream_items(run_dir) .into_iter() .any(|item| item.name() == Some(event_name)) } @@ -160,7 +160,7 @@ fn run_state(run_dir: &Path) -> RunProjection { )) } -fn run_events(run_dir: &Path) -> Vec { +fn run_stream_items(run_dir: &Path) -> Vec { let run_id = infer_run_id(run_dir); let runs_dir = run_dir.parent().expect("run dir should have parent"); let storage_dir = runs_dir.parent().expect("runs dir should have parent"); diff --git a/lib/packages/fabro-api-client/src/models/run-checkpoint.ts b/lib/packages/fabro-api-client/src/models/run-checkpoint.ts index 1d6481b7e..adf3fde23 100644 --- a/lib/packages/fabro-api-client/src/models/run-checkpoint.ts +++ b/lib/packages/fabro-api-client/src/models/run-checkpoint.ts @@ -15,47 +15,19 @@ /** - * Serializable snapshot of execution state for crash recovery and resume. + * A checkpoint Fabro recorded for the run: when, at which node, and the commit the workspace was checkpointed at, when it was committed. */ export interface RunCheckpoint { /** - * ISO 8601 timestamp when the checkpoint was created. + * When the checkpoint was recorded. */ 'timestamp': string; /** - * Identifier of the node being executed at checkpoint time. + * The node the checkpoint was recorded for. */ 'current_node': string; - /** - * Identifiers of nodes that have completed execution. - */ - 'completed_nodes': Array; - /** - * Map of node identifier to retry count. - */ - 'node_retries': { [key: string]: number; }; - /** - * Key-value context map accumulated during execution. - */ - 'context_values': { [key: string]: any; }; - /** - * Map of node identifier to outcome data for goal gate checks after resume. - */ - 'node_outcomes'?: { [key: string]: any; }; - /** - * The node to resume execution at after this checkpoint. - */ - 'next_node_id'?: string; /** * SHA of the git commit created at this checkpoint. */ 'git_commit_sha'?: string; - /** - * Failure signature counts within the main loop. - */ - 'loop_failure_signatures'?: { [key: string]: any; }; - /** - * Failure signature counts across loop_restart edges. - */ - 'restart_failure_signatures'?: { [key: string]: any; }; }