commit 8ae35cd8a0ca85b2fbb8c4e5570680fb8ddbc339 Author: Fabro Date: Mon May 4 16:07:39 2026 -0400 init run ⚒️ Generated with [Fabro](https://fabro.sh) diff --git a/graph.fabro b/graph.fabro new file mode 100644 index 000000000..bfd5da463 --- /dev/null +++ b/graph.fabro @@ -0,0 +1,37 @@ +digraph ImplementPlan { + graph [ + goal="Implement and simplify", + model_stylesheet=" + * { model: claude-opus-4-7; } + " + ] + rankdir=LR + + start [shape=Mdiamond, label="Start"] + exit [shape=Msquare, label="Exit"] + + toolchain [label="Toolchain", shape=parallelogram, script="command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", max_retries=0] + preflight_compile [label="Preflight Compile", shape=parallelogram, script="cargo check -q --workspace 2>&1", max_retries=0] + preflight_lint [label="Preflight Lint", shape=parallelogram, script="cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", max_retries=0] + fix_lints [label="Fix Lints", prompt="The preflight lint step failed. Read the build output from context and fix all clippy lint warnings.", max_visits=3] + implement [label="Implement", prompt="Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD."] + simplify_opus [label="Simplify (Opus)", prompt="@prompts/simplify.md"] + simplify_gpt [label="Simplify (GPT-55)", prompt="@prompts/simplify.md", model="gpt-55"] + verify [label="Verify", shape=parallelogram, script="cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1", goal_gate=true, retry_target="fixup"] + fixup [label="Fixup", prompt="The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors.", max_visits=3] + fmt [label="Format", shape=parallelogram, script="cargo +nightly-2026-04-14 fmt --all 2>&1", max_retries=0] + + start -> toolchain + toolchain -> preflight_compile [condition="outcome=succeeded"] + toolchain -> exit + preflight_compile -> preflight_lint [condition="outcome=succeeded"] + preflight_compile -> exit + preflight_lint -> implement [condition="outcome=succeeded"] + preflight_lint -> fix_lints + fix_lints -> preflight_lint + implement -> simplify_opus -> simplify_gpt -> verify + verify -> fmt [condition="outcome=succeeded"] + verify -> fixup + fixup -> verify + fmt -> exit +} diff --git a/run.json b/run.json new file mode 100644 index 000000000..90af3cc5a --- /dev/null +++ b/run.json @@ -0,0 +1,521 @@ +{ + "spec": { + "run_id": "01KQT9MH7PZ2T0694NH0YFQ6Q9", + "settings": { + "project": { + "name": null, + "description": null, + "directory": ".", + "metadata": {} + }, + "workflow": { + "name": null, + "description": null, + "graph": "workflow.fabro", + "metadata": {} + }, + "run": { + "goal": { + "type": "inline", + "value": "# Billing & Stages: Read From Projection\n\n## Context\n\nThe Billing tab on a running run omits the in-flight stage entirely, and the footer total runtime is frozen at the last server response.\n\nRoot cause: `GET /runs/{id}/billing` and `GET /runs/{id}/stages` (both in `lib/crates/fabro-server/src/server/handler/billing.rs`) bypass `RunProjection` and read `checkpoint.completed_nodes` + `checkpoint.node_outcomes` directly. The checkpoint only knows about *finished* nodes, so in-flight stages are invisible. `list_run_stages` had to grow a `next_node_id` workaround at `:113`; billing has no equivalent.\n\n`RunProjection` is the canonical event-sourced read model. `StageStarted` already creates a `StageProjection` entry the moment a stage begins (`run_state.rs:289`). The projection just doesn't yet store `started_at`, completion duration, billing usage, or `state` (Retrying vs Running).\n\nGoal: extend `StageProjection` with the missing event-derived fields, then collapse both handlers to thin views over `RunProjection.iter_stages()`. In-flight rows fall out for free. The frontend ticks runtime client-side using a server-supplied `started_at`.\n\nAudit confirmed these are the only two read endpoints with the bypass pattern.\n\n## Plan\n\n### 1. Extend `StageProjection`\n\nFile: `lib/crates/fabro-types/src/run_projection.rs`\n\nAdd four fields to `StageProjection`:\n\n```rust\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub started_at: Option>,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub duration_ms: Option,\n#[serde(skip)] // server-internal; not on the wire\npub usage: Option,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub state: Option,\n```\n\nWhy store `state` instead of deriving: the reducer needs to track `Retrying` (from `StageRetrying` events), which is not derivable from `completion` alone. Storing the field keeps the projection correct and removes the need for the existing `active_stage_state_from_events` event-replay (`billing.rs:19`). Use `Option<_>` so old serialized projections deserialize as `None` and can fall through a derivation helper.\n\nWhy `usage` is `#[serde(skip)]`: `BilledModelUsage` has no OpenAPI schema today (only `BilledTokenCounts` does, at `fabro-api.yaml:5756`). Modeling the full nested usage shape is out of scope for this PR, and `/runs/{id}/state` consumers can hit `/billing` if they need per-stage tokens. The billing handler reads `stage.usage` in-process to build `RunBillingStage.billing`. The field still survives in-process projection rebuild because `apply_event` reapplies it from `StageCompletedProps.billing` on every load.\n\nHelper methods:\n\n```rust\npub fn effective_state(&self) -> StageState {\n self.state.unwrap_or_else(|| match &self.completion {\n Some(c) => StageState::from(c.outcome),\n None => StageState::Running,\n })\n}\n\npub fn runtime_secs(&self, now: DateTime) -> Option {\n // Live state ticks; only use stored duration_ms once terminal.\n // This handles retries safely: even if a previous failed attempt left\n // `duration_ms` set, the new `state = Running` makes us recompute live.\n let state = self.effective_state();\n if matches!(state, StageState::Running | StageState::Retrying | StageState::Pending) {\n return self.started_at.map(|started| {\n now.signed_duration_since(started)\n .num_milliseconds()\n .max(0) as f64\n / 1000.0\n });\n }\n self.duration_ms.map(|ms| ms as f64 / 1000.0)\n}\n```\n\n`effective_state` keeps old serialized projections working without a backfill.\n\nUpdate `StageProjection::new` to default the four new fields to `None`.\n\n### 2. Capture the new fields in the reducer\n\nFile: `lib/crates/fabro-store/src/run_state.rs`. The reducer already has `let ts = stored.ts` in scope at `:46`.\n\n- `StageStarted` arm (`:289`): add a `StageProjection::reset_for_new_attempt(&mut self)` helper and call it after `stage_entry(...)`, then set `stage.started_at = Some(ts)` and `stage.state = Some(StageState::Running)`.\n\n `reset_for_new_attempt` clears **every attempt-result field**, because all of them are repopulated by per-attempt lifecycle events (`run_state.rs:299, 306, 312, 324, 338, 344, 350, 359, 375`) and would otherwise leak prior-attempt data on retry:\n\n - `completion`, `duration_ms`, `usage`, `state` (terminal data)\n - `response`, `prompt`, `provider_used`, `diff` (LLM/agent attempt data)\n - `script_invocation`, `script_timing`, `parallel_results` (handler attempt data)\n - `stdout`, `stderr`, `stdout_bytes`, `stderr_bytes`, `streams_separated`, `live_streaming`, `termination` (command-output attempt data)\n\n The only fields preserved are `first_event_seq` (identity / sort key, set on first creation) and `started_at` / `state` which are written immediately after the reset. Without this reset, a retry with reused visit would leave `state = Running` alongside `completion.outcome = Failed` and prior `stdout`/`stderr` content — inconsistent projection state visible via `/runs/{id}/state`.\n- `StageCompleted` arm (`:312`): set `stage.duration_ms = Some(props.duration_ms)`, `stage.usage = props.billing.clone()`, `stage.state = Some(StageState::from(stage_outcome_from_props(props).status))`.\n- `StageFailed` arm (`:324`): set `stage.duration_ms = Some(props.duration_ms)` and `stage.state = Some(StageState::Failed)`.\n- `StageRetrying` arm: new — locate stage at current visit, set `stage.state = Some(StageState::Retrying)`. (No corresponding handler exists today.)\n\nAdd unit tests in the existing `#[cfg(test)] mod tests` block for each arm and one transition test (`StageStarted → StageFailed → StageRetrying → StageStarted` returns to `Running`).\n\n### 3. Rewrite `get_run_billing`\n\nFile: `lib/crates/fabro-server/src/server/handler/billing.rs:128`\n\nReplace the `checkpoint.completed_nodes` loop (`:179`) with:\n\n1. Load `RunProjection` once (already done at `:140`).\n2. Capture `now: DateTime` once.\n3. Collect `(StageId, &StageProjection)` from `projection.iter_stages()` into a `Vec`.\n4. Aggregate by `node_id` to align with finalized output (`fabro-workflow/src/pipeline/finalize.rs:113`):\n - **Order**: first occurrence wins. For each `node_id`, the sort key is the **minimum** `first_event_seq` across all of that node's visits (i.e. when the node first appeared in the event log).\n - **Data**: latest visit wins. The displayed row uses fields from the entry with the largest `visit` for that node_id.\n - This produces the same A, B order for an A→B→A loop that finalize produces. The current live handler iterates `checkpoint.completed_nodes: Vec` directly and could emit duplicate rows for revisits; the new behavior collapses them, intentionally matching finalize.\n5. Sort the deduped rows by the per-node_id minimum `first_event_seq` from step 4.\n6. For each stage, build a `RunBillingStage`:\n - `stage`: `BillingStageRef { id, name = node_id }`.\n - `model`: from `stage.usage.as_ref().map(|u| ModelReference { id: u.model_id().to_string() })`.\n - `billing`: from `stage.usage` via the existing `BilledTokenCounts` shape; default if `None`.\n - `runtime_secs`: `stage.runtime_secs(now).unwrap_or(0.0)`.\n - `started_at`: `stage.started_at` (new field — see §5).\n - `state`: `stage.effective_state()` (new field — see §5).\n7. Totals: server-side total `runtime_secs` sums all rendered row runtimes (now includes the in-flight row's elapsed time). Tokens & cost via `BilledTokenCounts::from_billed_usage` over completed-stage usage — same as today.\n8. By-model breakdown: same as today, built from projection-derived usage list.\n\nDrop the dependency on `fabro_workflow::extract_stage_durations_from_events` from this handler.\n\n### 4. Rewrite `list_run_stages`\n\nSame handler, `:38`.\n\nSame shape as §3 for `RunStage`:\n\n- Iterate `projection.iter_stages()`, dedupe by node_id with the same rule as §3 step 4: latest-visit data, sort by per-node_id minimum `first_event_seq`.\n- `RunStage { id, name, status: stage.effective_state(), duration_secs: stage.runtime_secs(now), dot_id: Some(node_id), started_at: stage.started_at }`.\n- Drop the `next_node_id` synthesis at `:113`.\n- Drop the live-vs-store fork at `:50–78`; the projection is updated as events are written, so a single `state.store.open_run_reader(...).state()` read suffices.\n- Delete `active_stage_state_from_events` at `:19` — no longer needed; `state` is on the projection.\n\n### 5. OpenAPI: extend three schemas\n\nFile: `docs/public/api-reference/fabro-api.yaml`\n\n- **`RunBillingStage`** (`:6610`): add optional `started_at: string (date-time)` and `state: $ref StageState`. Frontend uses `state` to detect in-flight rows.\n- **`RunStage`** (`:6316`): add optional `started_at: string (date-time)`. `status: StageState` already exists.\n- **`StageProjection`** (`:5279`): add optional `started_at`, `duration_ms`, and `state: StageState`. **Do not** add `usage` here — the field is `#[serde(skip)]` server-internal (see §1). `BilledModelUsage` is not currently an OpenAPI schema and modeling it would balloon this PR's surface; `/runs/{id}/state` consumers needing per-stage tokens hit `/billing` instead.\n\nAfter editing: `cargo build -p fabro-api` regenerates Rust types; `cd lib/packages/fabro-api-client && bun run generate` regenerates the TS client.\n\n### 6. Update demo fixtures\n\nFile: `lib/crates/fabro-server/src/demo/mod.rs`\n\n- `RunStage` literals at `:1184, 1191, 1198, 1205` — add `started_at: None`.\n- `RunBillingStage` literals at `:1233, 1252, 1271, 1290` — add `started_at: None` and `state: StageState::Succeeded` (or appropriate per fixture).\n- Any `StageProjection` literals in tests/fixtures — search `rg \"StageProjection \\{\"` and add the new optional fields (typically `..Default::default()` shape if used).\n\n### 7. Frontend: invalidate on stage events + live tick\n\nFiles: `apps/fabro-web/app/lib/run-events.ts`, `apps/fabro-web/app/routes/run-billing.tsx`.\n\n`run-events.ts`:\n- Add `\"stage.retrying\"` to the `STAGE_EVENTS` set at `:35`. The projection now stores Retrying state, so the UI must refetch when this event arrives.\n- Add `queryKeys.runs.billing(runId)` to the `STAGE_EVENTS` invalidation list at `:75`.\n- Update the `queryKeysForRunEvent` test in `run-events.test.tsx` to verify `stage.retrying` invalidates stages, billing, events, and (when stage_id present) stage turns.\n\n`run-billing.tsx`:\n- Detect in-flight via the new `state` field: `state === \"running\" || state === \"retrying\"`.\n- If any row is in-flight, run a `useEffect` `setInterval(..., 1000)` that bumps a `now` state. Render the in-flight row's runtime as `(now − new Date(started_at)) / 1000`.\n- **Footer total**: while ticking, derive total from the rendered row runtimes — sum up the displayed seconds (which now include the live elapsed for the in-flight row). Otherwise (terminal run) use `billing.totals.runtime_secs` from the server.\n- Drop the empty-state at `:83` when any in-flight row exists; the table appears as soon as the first stage starts.\n\nUpdate `apps/fabro-web/app/routes/run-billing.test.tsx`:\n- Extend fixtures with `started_at` and `state`.\n- Add a test for an in-flight row (state = `running`) that asserts (a) the row renders, (b) the footer total includes the elapsed time, (c) the table is shown even when no stage has completed.\n\n### 8. What stays out of scope\n\n- **Live tokens during a stage.** Requires a new `agent.turn.completed { usage }` event from `fabro-agent`/`fabro-llm` plus a reducer arm to accumulate onto `StageProjection.usage`. The schema in §1 is ready; instrumenting it is a separate change.\n- **Per-visit billing rows.** Today's behavior aggregates by node_id (latest visit). One row per retry/revisit is a UX decision separate from this fix.\n- **Removing `checkpoint.node_outcomes`.** Still used by workflow execution: `artifact.rs:92,134`, `finalize.rs:119,394`, retro/conditionals. Leave it.\n- **Mixed in-memory/projection reads on `/checkpoint` and `/graph`.** Different shape of issue; not this PR.\n\n### 9. API round-trip tests\n\nFiles: `lib/crates/fabro-api/tests/stage_projection_round_trip.rs`, `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs`.\n\nExtend the representative-JSON cases:\n\n- `stage_projection_round_trip.rs`: add `started_at`, `duration_ms`, `state` to the JSON fixture and assert they round-trip. Confirms the OpenAPI schema and Rust type stay in lock-step for the new fields.\n- `run_billing_stage_round_trip.rs`: add `started_at` and `state` to the JSON fixture and assert they round-trip. Add a second case for an in-flight row (`state = \"running\"`, no `model`, zero `billing`).\n\nThese prevent silent drift if the OpenAPI schema and Rust type ever diverge on the new fields.\n\n## Files to modify\n\n- `lib/crates/fabro-types/src/run_projection.rs` — fields + helpers\n- `lib/crates/fabro-store/src/run_state.rs` — reducer arms (incl. new `StageRetrying`) + tests\n- `lib/crates/fabro-server/src/server/handler/billing.rs` — both handlers rewritten; delete `active_stage_state_from_events`\n- `lib/crates/fabro-server/src/server/tests.rs` — keep `list_run_stages_projects_retrying_until_completion`; verify it still passes via the new projection-based path\n- `lib/crates/fabro-server/src/demo/mod.rs` — fixture updates\n- `docs/public/api-reference/fabro-api.yaml` — `RunBillingStage`, `RunStage`, `StageProjection`\n- `lib/packages/fabro-api-client` — regenerated\n- `apps/fabro-web/app/lib/run-events.ts` — billing invalidation on stage events\n- `apps/fabro-web/app/routes/run-billing.tsx` — in-flight detection + tick + derived footer total\n- `apps/fabro-web/app/routes/run-billing.test.tsx` — new fixtures + in-flight + footer-tick assertions\n- `lib/crates/fabro-api/tests/stage_projection_round_trip.rs` — extend fixture with new fields\n- `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs` — extend fixture with new fields, add in-flight case\n- `apps/fabro-web/app/lib/run-events.test.tsx` — assert `stage.retrying` invalidates billing/stages/events\n\n## Existing utilities to reuse\n\n- `RunProjection::iter_stages()` — `lib/crates/fabro-types/src/run_projection.rs:102`\n- `StageProjection::first_event_seq` — already a `NonZeroU32`, ready as sort key\n- `StageState` — `lib/crates/fabro-types/src/outcome.rs:111` with `From` already wired\n- `BilledTokenCounts::from_billed_usage` — used by current totals path\n- `accumulate_model_billing` — `lib/crates/fabro-server/src/server.rs:539`, used for by-model breakdown\n- chrono pattern: `now.signed_duration_since(...).num_milliseconds().max(0) as f64 / 1000.0` (e.g. `lib/crates/fabro-cli/src/commands/runs/list.rs:99`)\n\n## Verification\n\n1. **Reducer unit tests** in `run_state.rs`:\n - `stage_started_records_started_at_and_running_state`\n - `stage_completed_records_duration_usage_and_terminal_state`\n - `stage_failed_records_duration_and_failed_state`\n - `stage_retrying_sets_retrying_state`\n - `stage_started_after_retrying_returns_to_running` (transition)\n2. **Existing test must still pass**: `list_run_stages_projects_retrying_until_completion` (`server/tests.rs:2126`) — covers Retrying via the new projection path.\n3. **New handler integration tests** in `lib/crates/fabro-server/tests/it/scenario/usage.rs`:\n - **Mid-run snapshot**: pause workflow with one completed and one in-flight stage; assert `/billing` returns two rows; in-flight row has `state = \"running\"`, `model = null`, zero `billing` tokens, non-zero `runtime_secs`; totals include the in-flight runtime.\n - **Retried node, mid-retry**: StageStarted → StageFailed (duration_ms = 10) → StageRetrying → StageStarted (no completion yet); assert the row's `state = \"running\"` and `runtime_secs` reflects elapsed since the **second** StageStarted, not the failed attempt's 10ms. Pin the regression risk that motivated the `runtime_secs()` priority inversion.\n - **Retried node, succeeded**: same prefix → StageCompleted; assert one row per node_id (latest visit), state `Succeeded`, duration = final attempt's `duration_ms`.\n - **Revisited node (loop, multi-node)**: emit A completed → B completed → A revisited+completed (visit=2). Assert (a) two rows total, (b) order is A, B (matches `finalize.rs:113`), (c) A's row carries the latest visit's data (visit=2 duration/usage), not the first visit's. Pins both the dedupe rule and the ordering rule against future drift.\n4. **Frontend tests** — `run-billing.test.tsx`:\n - In-flight row renders with runtime > 0.\n - Footer total ticks while the in-flight row ticks.\n - Empty-state hidden when an in-flight row exists.\n5. **End-to-end smoke** — `fabro run repl`, open `/runs//billing` in dev:\n - In-flight stage row appears immediately on `stage.started`.\n - Runtime ticks once per second.\n - On `stage.completed`, row gets `duration_ms` + tokens; next stage's row appears.\n - Footer reflects live in-flight runtime.\n6. **Conformance** — `cargo nextest run -p fabro-server`, `cd apps/fabro-web && bun run typecheck && bun test`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`. Run `cargo insta pending-snapshots` afterwards in case any snapshot tests pick up the new optional fields.\n\n## Unresolved questions\n\n- For runs with retried/revisited nodes, is \"latest visit per node_id\" the right billing display, or should we eventually expose all visits as separate rows? Plan matches current behavior; flagging for future.\n- `StageProjection.usage` is server-internal (`#[serde(skip)]`) for this PR. If a future consumer of `/runs/{id}/state` needs per-stage tokens, we'd model `BilledModelUsage` as an OpenAPI schema and unskip it — separate change.\n" + }, + "working_dir": null, + "metadata": {}, + "inputs": {}, + "model": { + "provider": "anthropic", + "name": "claude-sonnet-4-6", + "fallbacks": [] + }, + "git": { + "author": null + }, + "prepare": { + "commands": [], + "timeout_ms": 300000 + }, + "execution": { + "mode": "normal", + "approval": "prompt", + "retros": true + }, + "checkpoint": { + "exclude_globs": [] + }, + "sandbox": { + "provider": "daytona", + "preserve": false, + "devcontainer": false, + "env": {}, + "local": { + "worktree_mode": "always" + }, + "docker": { + "image": "buildpack-deps:noble", + "network_mode": null, + "memory_limit": 4000000000, + "cpu_quota": 200000, + "env_vars": {}, + "skip_clone": false + }, + "daytona": { + "auto_stop_interval": 30, + "labels": { + "repo": "fabro-sh/fabro" + }, + "snapshot": { + "name": "fabro-v8", + "cpu": 8, + "memory_gb": 16, + "disk_gb": 20, + "dockerfile": { + "type": "inline", + "value": "FROM ubuntu:24.04\n\nRUN apt-get update && apt-get install -y --no-install-recommends curl git ca-certificates build-essential pkg-config libssl-dev unzip python3 && rm -rf /var/lib/apt/lists/*\n\n# GitHub CLI\nRUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg | dd of=/usr/share/keyrings/githubcli-archive-keyring.gpg && echo \"deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main\" | tee /etc/apt/sources.list.d/github-cli.list > /dev/null && apt-get update && apt-get install -y --no-install-recommends gh && rm -rf /var/lib/apt/lists/*\n\n# Rust\nRUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y\nENV PATH=\"/root/.cargo/bin:${PATH}\"\nRUN rustup toolchain install nightly-2026-04-14 --profile minimal --component clippy,rustfmt\nRUN cargo install cargo-nextest --locked\nENV CARGO_INCREMENTAL=0\n\n# Bun\nRUN curl -fsSL https://bun.sh/install | bash\nENV PATH=\"/root/.bun/bin:${PATH}\"\n\nWORKDIR /root\n" + } + }, + "network": null, + "skip_clone": false + } + }, + "notifications": {}, + "interviews": { + "provider": null, + "slack": null, + "discord": null, + "teams": null + }, + "agent": { + "permissions": null, + "mcps": {} + }, + "hooks": [], + "scm": { + "provider": null, + "owner": null, + "repository": null, + "github": null + }, + "pull_request": { + "enabled": true, + "draft": false, + "auto_merge": false, + "merge_strategy": "squash" + }, + "artifacts": { + "include": [] + } + } + }, + "graph": { + "name": "ImplementPlan", + "nodes": { + "simplify_gpt": { + "id": "simplify_gpt", + "attrs": { + "provider": { + "String": "openai" + }, + "model": { + "String": "gpt-5.5" + }, + "label": { + "String": "Simplify (GPT-55)" + }, + "prompt": { + "String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)." + } + } + }, + "exit": { + "id": "exit", + "attrs": { + "shape": { + "String": "Msquare" + }, + "model": { + "String": "claude-opus-4-7" + }, + "label": { + "String": "Exit" + }, + "provider": { + "String": "anthropic" + } + } + }, + "implement": { + "id": "implement", + "attrs": { + "label": { + "String": "Implement" + }, + "model": { + "String": "claude-opus-4-7" + }, + "prompt": { + "String": "Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD." + }, + "provider": { + "String": "anthropic" + } + } + }, + "fmt": { + "id": "fmt", + "attrs": { + "script": { + "String": "cargo +nightly-2026-04-14 fmt --all 2>&1" + }, + "model": { + "String": "claude-opus-4-7" + }, + "shape": { + "String": "parallelogram" + }, + "provider": { + "String": "anthropic" + }, + "max_retries": { + "Integer": 0 + }, + "label": { + "String": "Format" + } + } + }, + "fixup": { + "id": "fixup", + "attrs": { + "prompt": { + "String": "The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors." + }, + "provider": { + "String": "anthropic" + }, + "max_visits": { + "Integer": 3 + }, + "model": { + "String": "claude-opus-4-7" + }, + "label": { + "String": "Fixup" + } + } + }, + "simplify_opus": { + "id": "simplify_opus", + "attrs": { + "model": { + "String": "claude-opus-4-7" + }, + "prompt": { + "String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)." + }, + "label": { + "String": "Simplify (Opus)" + }, + "provider": { + "String": "anthropic" + } + } + }, + "verify": { + "id": "verify", + "attrs": { + "provider": { + "String": "anthropic" + }, + "model": { + "String": "claude-opus-4-7" + }, + "shape": { + "String": "parallelogram" + }, + "retry_target": { + "String": "fixup" + }, + "goal_gate": { + "Boolean": true + }, + "script": { + "String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1" + }, + "label": { + "String": "Verify" + } + } + }, + "preflight_compile": { + "id": "preflight_compile", + "attrs": { + "label": { + "String": "Preflight Compile" + }, + "script": { + "String": "cargo check -q --workspace 2>&1" + }, + "shape": { + "String": "parallelogram" + }, + "max_retries": { + "Integer": 0 + }, + "model": { + "String": "claude-opus-4-7" + }, + "provider": { + "String": "anthropic" + } + } + }, + "preflight_lint": { + "id": "preflight_lint", + "attrs": { + "shape": { + "String": "parallelogram" + }, + "model": { + "String": "claude-opus-4-7" + }, + "label": { + "String": "Preflight Lint" + }, + "provider": { + "String": "anthropic" + }, + "script": { + "String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1" + }, + "max_retries": { + "Integer": 0 + } + } + }, + "toolchain": { + "id": "toolchain", + "attrs": { + "max_retries": { + "Integer": 0 + }, + "script": { + "String": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1" + }, + "shape": { + "String": "parallelogram" + }, + "provider": { + "String": "anthropic" + }, + "model": { + "String": "claude-opus-4-7" + }, + "label": { + "String": "Toolchain" + } + } + }, + "start": { + "id": "start", + "attrs": { + "shape": { + "String": "Mdiamond" + }, + "provider": { + "String": "anthropic" + }, + "model": { + "String": "claude-opus-4-7" + }, + "label": { + "String": "Start" + } + } + }, + "fix_lints": { + "id": "fix_lints", + "attrs": { + "max_visits": { + "Integer": 3 + }, + "prompt": { + "String": "The preflight lint step failed. Read the build output from context and fix all clippy lint warnings." + }, + "label": { + "String": "Fix Lints" + }, + "provider": { + "String": "anthropic" + }, + "model": { + "String": "claude-opus-4-7" + } + } + } + }, + "edges": [ + { + "from": "start", + "to": "toolchain", + "attrs": {} + }, + { + "from": "toolchain", + "to": "preflight_compile", + "attrs": { + "condition": { + "String": "outcome=succeeded" + } + } + }, + { + "from": "toolchain", + "to": "exit", + "attrs": {} + }, + { + "from": "preflight_compile", + "to": "preflight_lint", + "attrs": { + "condition": { + "String": "outcome=succeeded" + } + } + }, + { + "from": "preflight_compile", + "to": "exit", + "attrs": {} + }, + { + "from": "preflight_lint", + "to": "implement", + "attrs": { + "condition": { + "String": "outcome=succeeded" + } + } + }, + { + "from": "preflight_lint", + "to": "fix_lints", + "attrs": {} + }, + { + "from": "fix_lints", + "to": "preflight_lint", + "attrs": {} + }, + { + "from": "implement", + "to": "simplify_opus", + "attrs": {} + }, + { + "from": "simplify_opus", + "to": "simplify_gpt", + "attrs": {} + }, + { + "from": "simplify_gpt", + "to": "verify", + "attrs": {} + }, + { + "from": "verify", + "to": "fmt", + "attrs": { + "condition": { + "String": "outcome=succeeded" + } + } + }, + { + "from": "verify", + "to": "fixup", + "attrs": {} + }, + { + "from": "fixup", + "to": "verify", + "attrs": {} + }, + { + "from": "fmt", + "to": "exit", + "attrs": {} + } + ], + "attrs": { + "rankdir": { + "String": "LR" + }, + "goal": { + "String": "# Billing & Stages: Read From Projection\n\n## Context\n\nThe Billing tab on a running run omits the in-flight stage entirely, and the footer total runtime is frozen at the last server response.\n\nRoot cause: `GET /runs/{id}/billing` and `GET /runs/{id}/stages` (both in `lib/crates/fabro-server/src/server/handler/billing.rs`) bypass `RunProjection` and read `checkpoint.completed_nodes` + `checkpoint.node_outcomes` directly. The checkpoint only knows about *finished* nodes, so in-flight stages are invisible. `list_run_stages` had to grow a `next_node_id` workaround at `:113`; billing has no equivalent.\n\n`RunProjection` is the canonical event-sourced read model. `StageStarted` already creates a `StageProjection` entry the moment a stage begins (`run_state.rs:289`). The projection just doesn't yet store `started_at`, completion duration, billing usage, or `state` (Retrying vs Running).\n\nGoal: extend `StageProjection` with the missing event-derived fields, then collapse both handlers to thin views over `RunProjection.iter_stages()`. In-flight rows fall out for free. The frontend ticks runtime client-side using a server-supplied `started_at`.\n\nAudit confirmed these are the only two read endpoints with the bypass pattern.\n\n## Plan\n\n### 1. Extend `StageProjection`\n\nFile: `lib/crates/fabro-types/src/run_projection.rs`\n\nAdd four fields to `StageProjection`:\n\n```rust\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub started_at: Option>,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub duration_ms: Option,\n#[serde(skip)] // server-internal; not on the wire\npub usage: Option,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub state: Option,\n```\n\nWhy store `state` instead of deriving: the reducer needs to track `Retrying` (from `StageRetrying` events), which is not derivable from `completion` alone. Storing the field keeps the projection correct and removes the need for the existing `active_stage_state_from_events` event-replay (`billing.rs:19`). Use `Option<_>` so old serialized projections deserialize as `None` and can fall through a derivation helper.\n\nWhy `usage` is `#[serde(skip)]`: `BilledModelUsage` has no OpenAPI schema today (only `BilledTokenCounts` does, at `fabro-api.yaml:5756`). Modeling the full nested usage shape is out of scope for this PR, and `/runs/{id}/state` consumers can hit `/billing` if they need per-stage tokens. The billing handler reads `stage.usage` in-process to build `RunBillingStage.billing`. The field still survives in-process projection rebuild because `apply_event` reapplies it from `StageCompletedProps.billing` on every load.\n\nHelper methods:\n\n```rust\npub fn effective_state(&self) -> StageState {\n self.state.unwrap_or_else(|| match &self.completion {\n Some(c) => StageState::from(c.outcome),\n None => StageState::Running,\n })\n}\n\npub fn runtime_secs(&self, now: DateTime) -> Option {\n // Live state ticks; only use stored duration_ms once terminal.\n // This handles retries safely: even if a previous failed attempt left\n // `duration_ms` set, the new `state = Running` makes us recompute live.\n let state = self.effective_state();\n if matches!(state, StageState::Running | StageState::Retrying | StageState::Pending) {\n return self.started_at.map(|started| {\n now.signed_duration_since(started)\n .num_milliseconds()\n .max(0) as f64\n / 1000.0\n });\n }\n self.duration_ms.map(|ms| ms as f64 / 1000.0)\n}\n```\n\n`effective_state` keeps old serialized projections working without a backfill.\n\nUpdate `StageProjection::new` to default the four new fields to `None`.\n\n### 2. Capture the new fields in the reducer\n\nFile: `lib/crates/fabro-store/src/run_state.rs`. The reducer already has `let ts = stored.ts` in scope at `:46`.\n\n- `StageStarted` arm (`:289`): add a `StageProjection::reset_for_new_attempt(&mut self)` helper and call it after `stage_entry(...)`, then set `stage.started_at = Some(ts)` and `stage.state = Some(StageState::Running)`.\n\n `reset_for_new_attempt` clears **every attempt-result field**, because all of them are repopulated by per-attempt lifecycle events (`run_state.rs:299, 306, 312, 324, 338, 344, 350, 359, 375`) and would otherwise leak prior-attempt data on retry:\n\n - `completion`, `duration_ms`, `usage`, `state` (terminal data)\n - `response`, `prompt`, `provider_used`, `diff` (LLM/agent attempt data)\n - `script_invocation`, `script_timing`, `parallel_results` (handler attempt data)\n - `stdout`, `stderr`, `stdout_bytes`, `stderr_bytes`, `streams_separated`, `live_streaming`, `termination` (command-output attempt data)\n\n The only fields preserved are `first_event_seq` (identity / sort key, set on first creation) and `started_at` / `state` which are written immediately after the reset. Without this reset, a retry with reused visit would leave `state = Running` alongside `completion.outcome = Failed` and prior `stdout`/`stderr` content — inconsistent projection state visible via `/runs/{id}/state`.\n- `StageCompleted` arm (`:312`): set `stage.duration_ms = Some(props.duration_ms)`, `stage.usage = props.billing.clone()`, `stage.state = Some(StageState::from(stage_outcome_from_props(props).status))`.\n- `StageFailed` arm (`:324`): set `stage.duration_ms = Some(props.duration_ms)` and `stage.state = Some(StageState::Failed)`.\n- `StageRetrying` arm: new — locate stage at current visit, set `stage.state = Some(StageState::Retrying)`. (No corresponding handler exists today.)\n\nAdd unit tests in the existing `#[cfg(test)] mod tests` block for each arm and one transition test (`StageStarted → StageFailed → StageRetrying → StageStarted` returns to `Running`).\n\n### 3. Rewrite `get_run_billing`\n\nFile: `lib/crates/fabro-server/src/server/handler/billing.rs:128`\n\nReplace the `checkpoint.completed_nodes` loop (`:179`) with:\n\n1. Load `RunProjection` once (already done at `:140`).\n2. Capture `now: DateTime` once.\n3. Collect `(StageId, &StageProjection)` from `projection.iter_stages()` into a `Vec`.\n4. Aggregate by `node_id` to align with finalized output (`fabro-workflow/src/pipeline/finalize.rs:113`):\n - **Order**: first occurrence wins. For each `node_id`, the sort key is the **minimum** `first_event_seq` across all of that node's visits (i.e. when the node first appeared in the event log).\n - **Data**: latest visit wins. The displayed row uses fields from the entry with the largest `visit` for that node_id.\n - This produces the same A, B order for an A→B→A loop that finalize produces. The current live handler iterates `checkpoint.completed_nodes: Vec` directly and could emit duplicate rows for revisits; the new behavior collapses them, intentionally matching finalize.\n5. Sort the deduped rows by the per-node_id minimum `first_event_seq` from step 4.\n6. For each stage, build a `RunBillingStage`:\n - `stage`: `BillingStageRef { id, name = node_id }`.\n - `model`: from `stage.usage.as_ref().map(|u| ModelReference { id: u.model_id().to_string() })`.\n - `billing`: from `stage.usage` via the existing `BilledTokenCounts` shape; default if `None`.\n - `runtime_secs`: `stage.runtime_secs(now).unwrap_or(0.0)`.\n - `started_at`: `stage.started_at` (new field — see §5).\n - `state`: `stage.effective_state()` (new field — see §5).\n7. Totals: server-side total `runtime_secs` sums all rendered row runtimes (now includes the in-flight row's elapsed time). Tokens & cost via `BilledTokenCounts::from_billed_usage` over completed-stage usage — same as today.\n8. By-model breakdown: same as today, built from projection-derived usage list.\n\nDrop the dependency on `fabro_workflow::extract_stage_durations_from_events` from this handler.\n\n### 4. Rewrite `list_run_stages`\n\nSame handler, `:38`.\n\nSame shape as §3 for `RunStage`:\n\n- Iterate `projection.iter_stages()`, dedupe by node_id with the same rule as §3 step 4: latest-visit data, sort by per-node_id minimum `first_event_seq`.\n- `RunStage { id, name, status: stage.effective_state(), duration_secs: stage.runtime_secs(now), dot_id: Some(node_id), started_at: stage.started_at }`.\n- Drop the `next_node_id` synthesis at `:113`.\n- Drop the live-vs-store fork at `:50–78`; the projection is updated as events are written, so a single `state.store.open_run_reader(...).state()` read suffices.\n- Delete `active_stage_state_from_events` at `:19` — no longer needed; `state` is on the projection.\n\n### 5. OpenAPI: extend three schemas\n\nFile: `docs/public/api-reference/fabro-api.yaml`\n\n- **`RunBillingStage`** (`:6610`): add optional `started_at: string (date-time)` and `state: $ref StageState`. Frontend uses `state` to detect in-flight rows.\n- **`RunStage`** (`:6316`): add optional `started_at: string (date-time)`. `status: StageState` already exists.\n- **`StageProjection`** (`:5279`): add optional `started_at`, `duration_ms`, and `state: StageState`. **Do not** add `usage` here — the field is `#[serde(skip)]` server-internal (see §1). `BilledModelUsage` is not currently an OpenAPI schema and modeling it would balloon this PR's surface; `/runs/{id}/state` consumers needing per-stage tokens hit `/billing` instead.\n\nAfter editing: `cargo build -p fabro-api` regenerates Rust types; `cd lib/packages/fabro-api-client && bun run generate` regenerates the TS client.\n\n### 6. Update demo fixtures\n\nFile: `lib/crates/fabro-server/src/demo/mod.rs`\n\n- `RunStage` literals at `:1184, 1191, 1198, 1205` — add `started_at: None`.\n- `RunBillingStage` literals at `:1233, 1252, 1271, 1290` — add `started_at: None` and `state: StageState::Succeeded` (or appropriate per fixture).\n- Any `StageProjection` literals in tests/fixtures — search `rg \"StageProjection \\{\"` and add the new optional fields (typically `..Default::default()` shape if used).\n\n### 7. Frontend: invalidate on stage events + live tick\n\nFiles: `apps/fabro-web/app/lib/run-events.ts`, `apps/fabro-web/app/routes/run-billing.tsx`.\n\n`run-events.ts`:\n- Add `\"stage.retrying\"` to the `STAGE_EVENTS` set at `:35`. The projection now stores Retrying state, so the UI must refetch when this event arrives.\n- Add `queryKeys.runs.billing(runId)` to the `STAGE_EVENTS` invalidation list at `:75`.\n- Update the `queryKeysForRunEvent` test in `run-events.test.tsx` to verify `stage.retrying` invalidates stages, billing, events, and (when stage_id present) stage turns.\n\n`run-billing.tsx`:\n- Detect in-flight via the new `state` field: `state === \"running\" || state === \"retrying\"`.\n- If any row is in-flight, run a `useEffect` `setInterval(..., 1000)` that bumps a `now` state. Render the in-flight row's runtime as `(now − new Date(started_at)) / 1000`.\n- **Footer total**: while ticking, derive total from the rendered row runtimes — sum up the displayed seconds (which now include the live elapsed for the in-flight row). Otherwise (terminal run) use `billing.totals.runtime_secs` from the server.\n- Drop the empty-state at `:83` when any in-flight row exists; the table appears as soon as the first stage starts.\n\nUpdate `apps/fabro-web/app/routes/run-billing.test.tsx`:\n- Extend fixtures with `started_at` and `state`.\n- Add a test for an in-flight row (state = `running`) that asserts (a) the row renders, (b) the footer total includes the elapsed time, (c) the table is shown even when no stage has completed.\n\n### 8. What stays out of scope\n\n- **Live tokens during a stage.** Requires a new `agent.turn.completed { usage }` event from `fabro-agent`/`fabro-llm` plus a reducer arm to accumulate onto `StageProjection.usage`. The schema in §1 is ready; instrumenting it is a separate change.\n- **Per-visit billing rows.** Today's behavior aggregates by node_id (latest visit). One row per retry/revisit is a UX decision separate from this fix.\n- **Removing `checkpoint.node_outcomes`.** Still used by workflow execution: `artifact.rs:92,134`, `finalize.rs:119,394`, retro/conditionals. Leave it.\n- **Mixed in-memory/projection reads on `/checkpoint` and `/graph`.** Different shape of issue; not this PR.\n\n### 9. API round-trip tests\n\nFiles: `lib/crates/fabro-api/tests/stage_projection_round_trip.rs`, `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs`.\n\nExtend the representative-JSON cases:\n\n- `stage_projection_round_trip.rs`: add `started_at`, `duration_ms`, `state` to the JSON fixture and assert they round-trip. Confirms the OpenAPI schema and Rust type stay in lock-step for the new fields.\n- `run_billing_stage_round_trip.rs`: add `started_at` and `state` to the JSON fixture and assert they round-trip. Add a second case for an in-flight row (`state = \"running\"`, no `model`, zero `billing`).\n\nThese prevent silent drift if the OpenAPI schema and Rust type ever diverge on the new fields.\n\n## Files to modify\n\n- `lib/crates/fabro-types/src/run_projection.rs` — fields + helpers\n- `lib/crates/fabro-store/src/run_state.rs` — reducer arms (incl. new `StageRetrying`) + tests\n- `lib/crates/fabro-server/src/server/handler/billing.rs` — both handlers rewritten; delete `active_stage_state_from_events`\n- `lib/crates/fabro-server/src/server/tests.rs` — keep `list_run_stages_projects_retrying_until_completion`; verify it still passes via the new projection-based path\n- `lib/crates/fabro-server/src/demo/mod.rs` — fixture updates\n- `docs/public/api-reference/fabro-api.yaml` — `RunBillingStage`, `RunStage`, `StageProjection`\n- `lib/packages/fabro-api-client` — regenerated\n- `apps/fabro-web/app/lib/run-events.ts` — billing invalidation on stage events\n- `apps/fabro-web/app/routes/run-billing.tsx` — in-flight detection + tick + derived footer total\n- `apps/fabro-web/app/routes/run-billing.test.tsx` — new fixtures + in-flight + footer-tick assertions\n- `lib/crates/fabro-api/tests/stage_projection_round_trip.rs` — extend fixture with new fields\n- `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs` — extend fixture with new fields, add in-flight case\n- `apps/fabro-web/app/lib/run-events.test.tsx` — assert `stage.retrying` invalidates billing/stages/events\n\n## Existing utilities to reuse\n\n- `RunProjection::iter_stages()` — `lib/crates/fabro-types/src/run_projection.rs:102`\n- `StageProjection::first_event_seq` — already a `NonZeroU32`, ready as sort key\n- `StageState` — `lib/crates/fabro-types/src/outcome.rs:111` with `From` already wired\n- `BilledTokenCounts::from_billed_usage` — used by current totals path\n- `accumulate_model_billing` — `lib/crates/fabro-server/src/server.rs:539`, used for by-model breakdown\n- chrono pattern: `now.signed_duration_since(...).num_milliseconds().max(0) as f64 / 1000.0` (e.g. `lib/crates/fabro-cli/src/commands/runs/list.rs:99`)\n\n## Verification\n\n1. **Reducer unit tests** in `run_state.rs`:\n - `stage_started_records_started_at_and_running_state`\n - `stage_completed_records_duration_usage_and_terminal_state`\n - `stage_failed_records_duration_and_failed_state`\n - `stage_retrying_sets_retrying_state`\n - `stage_started_after_retrying_returns_to_running` (transition)\n2. **Existing test must still pass**: `list_run_stages_projects_retrying_until_completion` (`server/tests.rs:2126`) — covers Retrying via the new projection path.\n3. **New handler integration tests** in `lib/crates/fabro-server/tests/it/scenario/usage.rs`:\n - **Mid-run snapshot**: pause workflow with one completed and one in-flight stage; assert `/billing` returns two rows; in-flight row has `state = \"running\"`, `model = null`, zero `billing` tokens, non-zero `runtime_secs`; totals include the in-flight runtime.\n - **Retried node, mid-retry**: StageStarted → StageFailed (duration_ms = 10) → StageRetrying → StageStarted (no completion yet); assert the row's `state = \"running\"` and `runtime_secs` reflects elapsed since the **second** StageStarted, not the failed attempt's 10ms. Pin the regression risk that motivated the `runtime_secs()` priority inversion.\n - **Retried node, succeeded**: same prefix → StageCompleted; assert one row per node_id (latest visit), state `Succeeded`, duration = final attempt's `duration_ms`.\n - **Revisited node (loop, multi-node)**: emit A completed → B completed → A revisited+completed (visit=2). Assert (a) two rows total, (b) order is A, B (matches `finalize.rs:113`), (c) A's row carries the latest visit's data (visit=2 duration/usage), not the first visit's. Pins both the dedupe rule and the ordering rule against future drift.\n4. **Frontend tests** — `run-billing.test.tsx`:\n - In-flight row renders with runtime > 0.\n - Footer total ticks while the in-flight row ticks.\n - Empty-state hidden when an in-flight row exists.\n5. **End-to-end smoke** — `fabro run repl`, open `/runs//billing` in dev:\n - In-flight stage row appears immediately on `stage.started`.\n - Runtime ticks once per second.\n - On `stage.completed`, row gets `duration_ms` + tokens; next stage's row appears.\n - Footer reflects live in-flight runtime.\n6. **Conformance** — `cargo nextest run -p fabro-server`, `cd apps/fabro-web && bun run typecheck && bun test`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`. Run `cargo insta pending-snapshots` afterwards in case any snapshot tests pick up the new optional fields.\n\n## Unresolved questions\n\n- For runs with retried/revisited nodes, is \"latest visit per node_id\" the right billing display, or should we eventually expose all visits as separate rows? Plan matches current behavior; flagging for future.\n- `StageProjection.usage` is server-internal (`#[serde(skip)]`) for this PR. If a future consumer of `/runs/{id}/state` needs per-stage tokens, we'd model `BilledModelUsage` as an OpenAPI schema and unskip it — separate change.\n" + }, + "model_stylesheet": { + "String": "\n * { model: claude-opus-4-7; }\n " + } + } + }, + "workflow_slug": "implement-plan", + "source_directory": "/Users/bhelmkamp/p/fabro-sh/fabro", + "provenance": { + "server": { + "version": "0.223.0-nightly.0" + }, + "client": { + "user_agent": "fabro-cli/0.223.0-nightly.0", + "name": "fabro-cli", + "version": "0.223.0-nightly.0" + }, + "subject": { + "kind": "user", + "identity": { + "issuer": "https://github.com", + "subject": "19" + }, + "login": "brynary", + "auth_method": "github" + } + }, + "manifest_blob": "53e7b6a190106c2fdff3bb715f3dda8428c1866652c1484ee41a082bea7449d0", + "definition_blob": "791b2ce7454b6fff8fa26bea4af48533a25d0e56c4cecc44be1ea071680b4f9d", + "git": { + "origin_url": "https://github.com/fabro-sh/fabro", + "branch": "main", + "sha": "b5b08e78d389a1746aa52490efe5d66f26a07eaa", + "dirty": "dirty", + "push_outcome": { + "type": "not_attempted" + } + }, + "in_place": false + }, + "graph_source": "digraph ImplementPlan {\n graph [\n goal=\"Implement and simplify\",\n model_stylesheet=\"\n * { model: claude-opus-4-7; }\n \"\n ]\n rankdir=LR\n\n start [shape=Mdiamond, label=\"Start\"]\n exit [shape=Msquare, label=\"Exit\"]\n\n toolchain [label=\"Toolchain\", shape=parallelogram, script=\"command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1\", max_retries=0]\n preflight_compile [label=\"Preflight Compile\", shape=parallelogram, script=\"cargo check -q --workspace 2>&1\", max_retries=0]\n preflight_lint [label=\"Preflight Lint\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1\", max_retries=0]\n fix_lints [label=\"Fix Lints\", prompt=\"The preflight lint step failed. Read the build output from context and fix all clippy lint warnings.\", max_visits=3]\n implement [label=\"Implement\", prompt=\"Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD.\"]\n simplify_opus [label=\"Simplify (Opus)\", prompt=\"@prompts/simplify.md\"]\n simplify_gpt [label=\"Simplify (GPT-55)\", prompt=\"@prompts/simplify.md\", model=\"gpt-55\"]\n verify [label=\"Verify\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1\", goal_gate=true, retry_target=\"fixup\"]\n fixup [label=\"Fixup\", prompt=\"The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors.\", max_visits=3]\n fmt [label=\"Format\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 fmt --all 2>&1\", max_retries=0]\n\n start -> toolchain\n toolchain -> preflight_compile [condition=\"outcome=succeeded\"]\n toolchain -> exit\n preflight_compile -> preflight_lint [condition=\"outcome=succeeded\"]\n preflight_compile -> exit\n preflight_lint -> implement [condition=\"outcome=succeeded\"]\n preflight_lint -> fix_lints\n fix_lints -> preflight_lint\n implement -> simplify_opus -> simplify_gpt -> verify\n verify -> fmt [condition=\"outcome=succeeded\"]\n verify -> fixup\n fixup -> verify\n fmt -> exit\n}\n", + "start": null, + "status": { + "kind": "starting" + }, + "status_updated_at": "2026-05-04T20:07:24.185723Z", + "pending_control": null, + "checkpoint": null, + "checkpoints": [], + "conclusion": null, + "retro": null, + "retro_prompt": null, + "retro_response": null, + "sandbox": { + "provider": "daytona", + "working_directory": "/home/daytona/workspace", + "identifier": "fabro-01KQT9MH7PZ2T0694NH0YFQ6Q9", + "repo_cloned": true, + "clone_origin_url": "https://github.com/fabro-sh/fabro", + "clone_branch": "main" + }, + "final_patch": null, + "pull_request": null, + "superseded_by": null, + "pending_interviews": {}, + "stages": {} +} \ No newline at end of file