From 305c81d9751da3e763403235938676c97c6ac2de Mon Sep 17 00:00:00 2001 From: Fabro Date: Mon, 4 May 2026 16:09:49 -0400 Subject: [PATCH] =?UTF-8?q?checkpoint=20=E2=9A=92=EF=B8=8F=20Generated=20w?= =?UTF-8?q?ith=20[Fabro](https://fabro.sh)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- run.json | 124 ++++++++++++++++-- stages/002-toolchain@1/script_timing.json | 11 ++ stages/002-toolchain@1/status.json | 6 + stages/002-toolchain@1/stderr.log | 1 + stages/002-toolchain@1/stdout.log | 1 + .../script_invocation.json | 5 + 6 files changed, 137 insertions(+), 11 deletions(-) create mode 100644 stages/002-toolchain@1/script_timing.json create mode 100644 stages/002-toolchain@1/status.json create mode 100644 stages/002-toolchain@1/stderr.log create mode 100644 stages/002-toolchain@1/stdout.log create mode 100644 stages/003-preflight_compile@1/script_invocation.json diff --git a/run.json b/run.json index f22c0b5bb..52afea66b 100644 --- a/run.json +++ b/run.json @@ -505,11 +505,12 @@ "status_updated_at": "2026-05-04T20:07:38.626332Z", "pending_control": null, "checkpoint": { - "timestamp": "2026-05-04T20:07:42.232687Z", - "current_node": "toolchain", + "timestamp": "2026-05-04T20:09:49.037779Z", + "current_node": "preflight_compile", "completed_nodes": [ "start", - "toolchain" + "toolchain", + "preflight_compile" ], "node_retries": {}, "context_values": { @@ -519,16 +520,18 @@ "command.stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", "graph.goal": "# Billing & Stages: Read From Projection\n\n## Context\n\nThe Billing tab on a running run omits the in-flight stage entirely, and the footer total runtime is frozen at the last server response.\n\nRoot cause: `GET /runs/{id}/billing` and `GET /runs/{id}/stages` (both in `lib/crates/fabro-server/src/server/handler/billing.rs`) bypass `RunProjection` and read `checkpoint.completed_nodes` + `checkpoint.node_outcomes` directly. The checkpoint only knows about *finished* nodes, so in-flight stages are invisible. `list_run_stages` had to grow a `next_node_id` workaround at `:113`; billing has no equivalent.\n\n`RunProjection` is the canonical event-sourced read model. `StageStarted` already creates a `StageProjection` entry the moment a stage begins (`run_state.rs:289`). The projection just doesn't yet store `started_at`, completion duration, billing usage, or `state` (Retrying vs Running).\n\nGoal: extend `StageProjection` with the missing event-derived fields, then collapse both handlers to thin views over `RunProjection.iter_stages()`. In-flight rows fall out for free. The frontend ticks runtime client-side using a server-supplied `started_at`.\n\nAudit confirmed these are the only two read endpoints with the bypass pattern.\n\n## Plan\n\n### 1. Extend `StageProjection`\n\nFile: `lib/crates/fabro-types/src/run_projection.rs`\n\nAdd four fields to `StageProjection`:\n\n```rust\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub started_at: Option>,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub duration_ms: Option,\n#[serde(skip)] // server-internal; not on the wire\npub usage: Option,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub state: Option,\n```\n\nWhy store `state` instead of deriving: the reducer needs to track `Retrying` (from `StageRetrying` events), which is not derivable from `completion` alone. Storing the field keeps the projection correct and removes the need for the existing `active_stage_state_from_events` event-replay (`billing.rs:19`). Use `Option<_>` so old serialized projections deserialize as `None` and can fall through a derivation helper.\n\nWhy `usage` is `#[serde(skip)]`: `BilledModelUsage` has no OpenAPI schema today (only `BilledTokenCounts` does, at `fabro-api.yaml:5756`). Modeling the full nested usage shape is out of scope for this PR, and `/runs/{id}/state` consumers can hit `/billing` if they need per-stage tokens. The billing handler reads `stage.usage` in-process to build `RunBillingStage.billing`. The field still survives in-process projection rebuild because `apply_event` reapplies it from `StageCompletedProps.billing` on every load.\n\nHelper methods:\n\n```rust\npub fn effective_state(&self) -> StageState {\n self.state.unwrap_or_else(|| match &self.completion {\n Some(c) => StageState::from(c.outcome),\n None => StageState::Running,\n })\n}\n\npub fn runtime_secs(&self, now: DateTime) -> Option {\n // Live state ticks; only use stored duration_ms once terminal.\n // This handles retries safely: even if a previous failed attempt left\n // `duration_ms` set, the new `state = Running` makes us recompute live.\n let state = self.effective_state();\n if matches!(state, StageState::Running | StageState::Retrying | StageState::Pending) {\n return self.started_at.map(|started| {\n now.signed_duration_since(started)\n .num_milliseconds()\n .max(0) as f64\n / 1000.0\n });\n }\n self.duration_ms.map(|ms| ms as f64 / 1000.0)\n}\n```\n\n`effective_state` keeps old serialized projections working without a backfill.\n\nUpdate `StageProjection::new` to default the four new fields to `None`.\n\n### 2. Capture the new fields in the reducer\n\nFile: `lib/crates/fabro-store/src/run_state.rs`. The reducer already has `let ts = stored.ts` in scope at `:46`.\n\n- `StageStarted` arm (`:289`): add a `StageProjection::reset_for_new_attempt(&mut self)` helper and call it after `stage_entry(...)`, then set `stage.started_at = Some(ts)` and `stage.state = Some(StageState::Running)`.\n\n `reset_for_new_attempt` clears **every attempt-result field**, because all of them are repopulated by per-attempt lifecycle events (`run_state.rs:299, 306, 312, 324, 338, 344, 350, 359, 375`) and would otherwise leak prior-attempt data on retry:\n\n - `completion`, `duration_ms`, `usage`, `state` (terminal data)\n - `response`, `prompt`, `provider_used`, `diff` (LLM/agent attempt data)\n - `script_invocation`, `script_timing`, `parallel_results` (handler attempt data)\n - `stdout`, `stderr`, `stdout_bytes`, `stderr_bytes`, `streams_separated`, `live_streaming`, `termination` (command-output attempt data)\n\n The only fields preserved are `first_event_seq` (identity / sort key, set on first creation) and `started_at` / `state` which are written immediately after the reset. Without this reset, a retry with reused visit would leave `state = Running` alongside `completion.outcome = Failed` and prior `stdout`/`stderr` content — inconsistent projection state visible via `/runs/{id}/state`.\n- `StageCompleted` arm (`:312`): set `stage.duration_ms = Some(props.duration_ms)`, `stage.usage = props.billing.clone()`, `stage.state = Some(StageState::from(stage_outcome_from_props(props).status))`.\n- `StageFailed` arm (`:324`): set `stage.duration_ms = Some(props.duration_ms)` and `stage.state = Some(StageState::Failed)`.\n- `StageRetrying` arm: new — locate stage at current visit, set `stage.state = Some(StageState::Retrying)`. (No corresponding handler exists today.)\n\nAdd unit tests in the existing `#[cfg(test)] mod tests` block for each arm and one transition test (`StageStarted → StageFailed → StageRetrying → StageStarted` returns to `Running`).\n\n### 3. Rewrite `get_run_billing`\n\nFile: `lib/crates/fabro-server/src/server/handler/billing.rs:128`\n\nReplace the `checkpoint.completed_nodes` loop (`:179`) with:\n\n1. Load `RunProjection` once (already done at `:140`).\n2. Capture `now: DateTime` once.\n3. Collect `(StageId, &StageProjection)` from `projection.iter_stages()` into a `Vec`.\n4. Aggregate by `node_id` to align with finalized output (`fabro-workflow/src/pipeline/finalize.rs:113`):\n - **Order**: first occurrence wins. For each `node_id`, the sort key is the **minimum** `first_event_seq` across all of that node's visits (i.e. when the node first appeared in the event log).\n - **Data**: latest visit wins. The displayed row uses fields from the entry with the largest `visit` for that node_id.\n - This produces the same A, B order for an A→B→A loop that finalize produces. The current live handler iterates `checkpoint.completed_nodes: Vec` directly and could emit duplicate rows for revisits; the new behavior collapses them, intentionally matching finalize.\n5. Sort the deduped rows by the per-node_id minimum `first_event_seq` from step 4.\n6. For each stage, build a `RunBillingStage`:\n - `stage`: `BillingStageRef { id, name = node_id }`.\n - `model`: from `stage.usage.as_ref().map(|u| ModelReference { id: u.model_id().to_string() })`.\n - `billing`: from `stage.usage` via the existing `BilledTokenCounts` shape; default if `None`.\n - `runtime_secs`: `stage.runtime_secs(now).unwrap_or(0.0)`.\n - `started_at`: `stage.started_at` (new field — see §5).\n - `state`: `stage.effective_state()` (new field — see §5).\n7. Totals: server-side total `runtime_secs` sums all rendered row runtimes (now includes the in-flight row's elapsed time). Tokens & cost via `BilledTokenCounts::from_billed_usage` over completed-stage usage — same as today.\n8. By-model breakdown: same as today, built from projection-derived usage list.\n\nDrop the dependency on `fabro_workflow::extract_stage_durations_from_events` from this handler.\n\n### 4. Rewrite `list_run_stages`\n\nSame handler, `:38`.\n\nSame shape as §3 for `RunStage`:\n\n- Iterate `projection.iter_stages()`, dedupe by node_id with the same rule as §3 step 4: latest-visit data, sort by per-node_id minimum `first_event_seq`.\n- `RunStage { id, name, status: stage.effective_state(), duration_secs: stage.runtime_secs(now), dot_id: Some(node_id), started_at: stage.started_at }`.\n- Drop the `next_node_id` synthesis at `:113`.\n- Drop the live-vs-store fork at `:50–78`; the projection is updated as events are written, so a single `state.store.open_run_reader(...).state()` read suffices.\n- Delete `active_stage_state_from_events` at `:19` — no longer needed; `state` is on the projection.\n\n### 5. OpenAPI: extend three schemas\n\nFile: `docs/public/api-reference/fabro-api.yaml`\n\n- **`RunBillingStage`** (`:6610`): add optional `started_at: string (date-time)` and `state: $ref StageState`. Frontend uses `state` to detect in-flight rows.\n- **`RunStage`** (`:6316`): add optional `started_at: string (date-time)`. `status: StageState` already exists.\n- **`StageProjection`** (`:5279`): add optional `started_at`, `duration_ms`, and `state: StageState`. **Do not** add `usage` here — the field is `#[serde(skip)]` server-internal (see §1). `BilledModelUsage` is not currently an OpenAPI schema and modeling it would balloon this PR's surface; `/runs/{id}/state` consumers needing per-stage tokens hit `/billing` instead.\n\nAfter editing: `cargo build -p fabro-api` regenerates Rust types; `cd lib/packages/fabro-api-client && bun run generate` regenerates the TS client.\n\n### 6. Update demo fixtures\n\nFile: `lib/crates/fabro-server/src/demo/mod.rs`\n\n- `RunStage` literals at `:1184, 1191, 1198, 1205` — add `started_at: None`.\n- `RunBillingStage` literals at `:1233, 1252, 1271, 1290` — add `started_at: None` and `state: StageState::Succeeded` (or appropriate per fixture).\n- Any `StageProjection` literals in tests/fixtures — search `rg \"StageProjection \\{\"` and add the new optional fields (typically `..Default::default()` shape if used).\n\n### 7. Frontend: invalidate on stage events + live tick\n\nFiles: `apps/fabro-web/app/lib/run-events.ts`, `apps/fabro-web/app/routes/run-billing.tsx`.\n\n`run-events.ts`:\n- Add `\"stage.retrying\"` to the `STAGE_EVENTS` set at `:35`. The projection now stores Retrying state, so the UI must refetch when this event arrives.\n- Add `queryKeys.runs.billing(runId)` to the `STAGE_EVENTS` invalidation list at `:75`.\n- Update the `queryKeysForRunEvent` test in `run-events.test.tsx` to verify `stage.retrying` invalidates stages, billing, events, and (when stage_id present) stage turns.\n\n`run-billing.tsx`:\n- Detect in-flight via the new `state` field: `state === \"running\" || state === \"retrying\"`.\n- If any row is in-flight, run a `useEffect` `setInterval(..., 1000)` that bumps a `now` state. Render the in-flight row's runtime as `(now − new Date(started_at)) / 1000`.\n- **Footer total**: while ticking, derive total from the rendered row runtimes — sum up the displayed seconds (which now include the live elapsed for the in-flight row). Otherwise (terminal run) use `billing.totals.runtime_secs` from the server.\n- Drop the empty-state at `:83` when any in-flight row exists; the table appears as soon as the first stage starts.\n\nUpdate `apps/fabro-web/app/routes/run-billing.test.tsx`:\n- Extend fixtures with `started_at` and `state`.\n- Add a test for an in-flight row (state = `running`) that asserts (a) the row renders, (b) the footer total includes the elapsed time, (c) the table is shown even when no stage has completed.\n\n### 8. What stays out of scope\n\n- **Live tokens during a stage.** Requires a new `agent.turn.completed { usage }` event from `fabro-agent`/`fabro-llm` plus a reducer arm to accumulate onto `StageProjection.usage`. The schema in §1 is ready; instrumenting it is a separate change.\n- **Per-visit billing rows.** Today's behavior aggregates by node_id (latest visit). One row per retry/revisit is a UX decision separate from this fix.\n- **Removing `checkpoint.node_outcomes`.** Still used by workflow execution: `artifact.rs:92,134`, `finalize.rs:119,394`, retro/conditionals. Leave it.\n- **Mixed in-memory/projection reads on `/checkpoint` and `/graph`.** Different shape of issue; not this PR.\n\n### 9. API round-trip tests\n\nFiles: `lib/crates/fabro-api/tests/stage_projection_round_trip.rs`, `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs`.\n\nExtend the representative-JSON cases:\n\n- `stage_projection_round_trip.rs`: add `started_at`, `duration_ms`, `state` to the JSON fixture and assert they round-trip. Confirms the OpenAPI schema and Rust type stay in lock-step for the new fields.\n- `run_billing_stage_round_trip.rs`: add `started_at` and `state` to the JSON fixture and assert they round-trip. Add a second case for an in-flight row (`state = \"running\"`, no `model`, zero `billing`).\n\nThese prevent silent drift if the OpenAPI schema and Rust type ever diverge on the new fields.\n\n## Files to modify\n\n- `lib/crates/fabro-types/src/run_projection.rs` — fields + helpers\n- `lib/crates/fabro-store/src/run_state.rs` — reducer arms (incl. new `StageRetrying`) + tests\n- `lib/crates/fabro-server/src/server/handler/billing.rs` — both handlers rewritten; delete `active_stage_state_from_events`\n- `lib/crates/fabro-server/src/server/tests.rs` — keep `list_run_stages_projects_retrying_until_completion`; verify it still passes via the new projection-based path\n- `lib/crates/fabro-server/src/demo/mod.rs` — fixture updates\n- `docs/public/api-reference/fabro-api.yaml` — `RunBillingStage`, `RunStage`, `StageProjection`\n- `lib/packages/fabro-api-client` — regenerated\n- `apps/fabro-web/app/lib/run-events.ts` — billing invalidation on stage events\n- `apps/fabro-web/app/routes/run-billing.tsx` — in-flight detection + tick + derived footer total\n- `apps/fabro-web/app/routes/run-billing.test.tsx` — new fixtures + in-flight + footer-tick assertions\n- `lib/crates/fabro-api/tests/stage_projection_round_trip.rs` — extend fixture with new fields\n- `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs` — extend fixture with new fields, add in-flight case\n- `apps/fabro-web/app/lib/run-events.test.tsx` — assert `stage.retrying` invalidates billing/stages/events\n\n## Existing utilities to reuse\n\n- `RunProjection::iter_stages()` — `lib/crates/fabro-types/src/run_projection.rs:102`\n- `StageProjection::first_event_seq` — already a `NonZeroU32`, ready as sort key\n- `StageState` — `lib/crates/fabro-types/src/outcome.rs:111` with `From` already wired\n- `BilledTokenCounts::from_billed_usage` — used by current totals path\n- `accumulate_model_billing` — `lib/crates/fabro-server/src/server.rs:539`, used for by-model breakdown\n- chrono pattern: `now.signed_duration_since(...).num_milliseconds().max(0) as f64 / 1000.0` (e.g. `lib/crates/fabro-cli/src/commands/runs/list.rs:99`)\n\n## Verification\n\n1. **Reducer unit tests** in `run_state.rs`:\n - `stage_started_records_started_at_and_running_state`\n - `stage_completed_records_duration_usage_and_terminal_state`\n - `stage_failed_records_duration_and_failed_state`\n - `stage_retrying_sets_retrying_state`\n - `stage_started_after_retrying_returns_to_running` (transition)\n2. **Existing test must still pass**: `list_run_stages_projects_retrying_until_completion` (`server/tests.rs:2126`) — covers Retrying via the new projection path.\n3. **New handler integration tests** in `lib/crates/fabro-server/tests/it/scenario/usage.rs`:\n - **Mid-run snapshot**: pause workflow with one completed and one in-flight stage; assert `/billing` returns two rows; in-flight row has `state = \"running\"`, `model = null`, zero `billing` tokens, non-zero `runtime_secs`; totals include the in-flight runtime.\n - **Retried node, mid-retry**: StageStarted → StageFailed (duration_ms = 10) → StageRetrying → StageStarted (no completion yet); assert the row's `state = \"running\"` and `runtime_secs` reflects elapsed since the **second** StageStarted, not the failed attempt's 10ms. Pin the regression risk that motivated the `runtime_secs()` priority inversion.\n - **Retried node, succeeded**: same prefix → StageCompleted; assert one row per node_id (latest visit), state `Succeeded`, duration = final attempt's `duration_ms`.\n - **Revisited node (loop, multi-node)**: emit A completed → B completed → A revisited+completed (visit=2). Assert (a) two rows total, (b) order is A, B (matches `finalize.rs:113`), (c) A's row carries the latest visit's data (visit=2 duration/usage), not the first visit's. Pins both the dedupe rule and the ordering rule against future drift.\n4. **Frontend tests** — `run-billing.test.tsx`:\n - In-flight row renders with runtime > 0.\n - Footer total ticks while the in-flight row ticks.\n - Empty-state hidden when an in-flight row exists.\n5. **End-to-end smoke** — `fabro run repl`, open `/runs//billing` in dev:\n - In-flight stage row appears immediately on `stage.started`.\n - Runtime ticks once per second.\n - On `stage.completed`, row gets `duration_ms` + tokens; next stage's row appears.\n - Footer reflects live in-flight runtime.\n6. **Conformance** — `cargo nextest run -p fabro-server`, `cd apps/fabro-web && bun run typecheck && bun test`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`. Run `cargo insta pending-snapshots` afterwards in case any snapshot tests pick up the new optional fields.\n\n## Unresolved questions\n\n- For runs with retried/revisited nodes, is \"latest visit per node_id\" the right billing display, or should we eventually expose all visits as separate rows? Plan matches current behavior; flagging for future.\n- `StageProjection.usage` is server-internal (`#[serde(skip)]`) for this PR. If a future consumer of `/runs/{id}/state` needs per-stage tokens, we'd model `BilledModelUsage` as an OpenAPI schema and unskip it — separate change.\n", "internal.fidelity": "compact", - "current_node": "toolchain", + "current_node": "preflight_compile", "outcome": "succeeded", "failure_signature": "", "internal.node_visit_count": 1, - "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c", + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "thread.toolchain.current_node": "preflight_compile", + "internal.retry_count.preflight_compile": 0, "internal.run_id": "01KQT9MH7PZ2T0694NH0YFQ6Q9", "thread.start.current_node": "toolchain", "graph.rankdir": "LR", "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", - "internal.thread_id": "start", + "internal.thread_id": "toolchain", "failure_class": "" }, "node_outcomes": { @@ -544,12 +547,22 @@ }, "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", "usage": null + }, + "preflight_compile": { + "status": "succeeded", + "context_updates": { + "command.stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" + }, + "notes": "Script completed: cargo check -q --workspace 2>&1", + "usage": null } }, - "next_node_id": "preflight_compile", + "next_node_id": "preflight_lint", "node_visits": { "start": 1, - "toolchain": 1 + "toolchain": 1, + "preflight_compile": 1 } }, "checkpoints": [ @@ -588,6 +601,58 @@ "start": 1 } } + ], + [ + 27, + { + "timestamp": "2026-05-04T20:07:46.182460Z", + "current_node": "toolchain", + "completed_nodes": [ + "start", + "toolchain" + ], + "node_retries": {}, + "context_values": { + "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", + "internal.retry_count.toolchain": 0, + "internal.run_id": "01KQT9MH7PZ2T0694NH0YFQ6Q9", + "failure_class": "", + "outcome": "succeeded", + "failure_signature": "", + "graph.rankdir": "LR", + "graph.goal": "# Billing & Stages: Read From Projection\n\n## Context\n\nThe Billing tab on a running run omits the in-flight stage entirely, and the footer total runtime is frozen at the last server response.\n\nRoot cause: `GET /runs/{id}/billing` and `GET /runs/{id}/stages` (both in `lib/crates/fabro-server/src/server/handler/billing.rs`) bypass `RunProjection` and read `checkpoint.completed_nodes` + `checkpoint.node_outcomes` directly. The checkpoint only knows about *finished* nodes, so in-flight stages are invisible. `list_run_stages` had to grow a `next_node_id` workaround at `:113`; billing has no equivalent.\n\n`RunProjection` is the canonical event-sourced read model. `StageStarted` already creates a `StageProjection` entry the moment a stage begins (`run_state.rs:289`). The projection just doesn't yet store `started_at`, completion duration, billing usage, or `state` (Retrying vs Running).\n\nGoal: extend `StageProjection` with the missing event-derived fields, then collapse both handlers to thin views over `RunProjection.iter_stages()`. In-flight rows fall out for free. The frontend ticks runtime client-side using a server-supplied `started_at`.\n\nAudit confirmed these are the only two read endpoints with the bypass pattern.\n\n## Plan\n\n### 1. Extend `StageProjection`\n\nFile: `lib/crates/fabro-types/src/run_projection.rs`\n\nAdd four fields to `StageProjection`:\n\n```rust\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub started_at: Option>,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub duration_ms: Option,\n#[serde(skip)] // server-internal; not on the wire\npub usage: Option,\n#[serde(default, skip_serializing_if = \"Option::is_none\")]\npub state: Option,\n```\n\nWhy store `state` instead of deriving: the reducer needs to track `Retrying` (from `StageRetrying` events), which is not derivable from `completion` alone. Storing the field keeps the projection correct and removes the need for the existing `active_stage_state_from_events` event-replay (`billing.rs:19`). Use `Option<_>` so old serialized projections deserialize as `None` and can fall through a derivation helper.\n\nWhy `usage` is `#[serde(skip)]`: `BilledModelUsage` has no OpenAPI schema today (only `BilledTokenCounts` does, at `fabro-api.yaml:5756`). Modeling the full nested usage shape is out of scope for this PR, and `/runs/{id}/state` consumers can hit `/billing` if they need per-stage tokens. The billing handler reads `stage.usage` in-process to build `RunBillingStage.billing`. The field still survives in-process projection rebuild because `apply_event` reapplies it from `StageCompletedProps.billing` on every load.\n\nHelper methods:\n\n```rust\npub fn effective_state(&self) -> StageState {\n self.state.unwrap_or_else(|| match &self.completion {\n Some(c) => StageState::from(c.outcome),\n None => StageState::Running,\n })\n}\n\npub fn runtime_secs(&self, now: DateTime) -> Option {\n // Live state ticks; only use stored duration_ms once terminal.\n // This handles retries safely: even if a previous failed attempt left\n // `duration_ms` set, the new `state = Running` makes us recompute live.\n let state = self.effective_state();\n if matches!(state, StageState::Running | StageState::Retrying | StageState::Pending) {\n return self.started_at.map(|started| {\n now.signed_duration_since(started)\n .num_milliseconds()\n .max(0) as f64\n / 1000.0\n });\n }\n self.duration_ms.map(|ms| ms as f64 / 1000.0)\n}\n```\n\n`effective_state` keeps old serialized projections working without a backfill.\n\nUpdate `StageProjection::new` to default the four new fields to `None`.\n\n### 2. Capture the new fields in the reducer\n\nFile: `lib/crates/fabro-store/src/run_state.rs`. The reducer already has `let ts = stored.ts` in scope at `:46`.\n\n- `StageStarted` arm (`:289`): add a `StageProjection::reset_for_new_attempt(&mut self)` helper and call it after `stage_entry(...)`, then set `stage.started_at = Some(ts)` and `stage.state = Some(StageState::Running)`.\n\n `reset_for_new_attempt` clears **every attempt-result field**, because all of them are repopulated by per-attempt lifecycle events (`run_state.rs:299, 306, 312, 324, 338, 344, 350, 359, 375`) and would otherwise leak prior-attempt data on retry:\n\n - `completion`, `duration_ms`, `usage`, `state` (terminal data)\n - `response`, `prompt`, `provider_used`, `diff` (LLM/agent attempt data)\n - `script_invocation`, `script_timing`, `parallel_results` (handler attempt data)\n - `stdout`, `stderr`, `stdout_bytes`, `stderr_bytes`, `streams_separated`, `live_streaming`, `termination` (command-output attempt data)\n\n The only fields preserved are `first_event_seq` (identity / sort key, set on first creation) and `started_at` / `state` which are written immediately after the reset. Without this reset, a retry with reused visit would leave `state = Running` alongside `completion.outcome = Failed` and prior `stdout`/`stderr` content — inconsistent projection state visible via `/runs/{id}/state`.\n- `StageCompleted` arm (`:312`): set `stage.duration_ms = Some(props.duration_ms)`, `stage.usage = props.billing.clone()`, `stage.state = Some(StageState::from(stage_outcome_from_props(props).status))`.\n- `StageFailed` arm (`:324`): set `stage.duration_ms = Some(props.duration_ms)` and `stage.state = Some(StageState::Failed)`.\n- `StageRetrying` arm: new — locate stage at current visit, set `stage.state = Some(StageState::Retrying)`. (No corresponding handler exists today.)\n\nAdd unit tests in the existing `#[cfg(test)] mod tests` block for each arm and one transition test (`StageStarted → StageFailed → StageRetrying → StageStarted` returns to `Running`).\n\n### 3. Rewrite `get_run_billing`\n\nFile: `lib/crates/fabro-server/src/server/handler/billing.rs:128`\n\nReplace the `checkpoint.completed_nodes` loop (`:179`) with:\n\n1. Load `RunProjection` once (already done at `:140`).\n2. Capture `now: DateTime` once.\n3. Collect `(StageId, &StageProjection)` from `projection.iter_stages()` into a `Vec`.\n4. Aggregate by `node_id` to align with finalized output (`fabro-workflow/src/pipeline/finalize.rs:113`):\n - **Order**: first occurrence wins. For each `node_id`, the sort key is the **minimum** `first_event_seq` across all of that node's visits (i.e. when the node first appeared in the event log).\n - **Data**: latest visit wins. The displayed row uses fields from the entry with the largest `visit` for that node_id.\n - This produces the same A, B order for an A→B→A loop that finalize produces. The current live handler iterates `checkpoint.completed_nodes: Vec` directly and could emit duplicate rows for revisits; the new behavior collapses them, intentionally matching finalize.\n5. Sort the deduped rows by the per-node_id minimum `first_event_seq` from step 4.\n6. For each stage, build a `RunBillingStage`:\n - `stage`: `BillingStageRef { id, name = node_id }`.\n - `model`: from `stage.usage.as_ref().map(|u| ModelReference { id: u.model_id().to_string() })`.\n - `billing`: from `stage.usage` via the existing `BilledTokenCounts` shape; default if `None`.\n - `runtime_secs`: `stage.runtime_secs(now).unwrap_or(0.0)`.\n - `started_at`: `stage.started_at` (new field — see §5).\n - `state`: `stage.effective_state()` (new field — see §5).\n7. Totals: server-side total `runtime_secs` sums all rendered row runtimes (now includes the in-flight row's elapsed time). Tokens & cost via `BilledTokenCounts::from_billed_usage` over completed-stage usage — same as today.\n8. By-model breakdown: same as today, built from projection-derived usage list.\n\nDrop the dependency on `fabro_workflow::extract_stage_durations_from_events` from this handler.\n\n### 4. Rewrite `list_run_stages`\n\nSame handler, `:38`.\n\nSame shape as §3 for `RunStage`:\n\n- Iterate `projection.iter_stages()`, dedupe by node_id with the same rule as §3 step 4: latest-visit data, sort by per-node_id minimum `first_event_seq`.\n- `RunStage { id, name, status: stage.effective_state(), duration_secs: stage.runtime_secs(now), dot_id: Some(node_id), started_at: stage.started_at }`.\n- Drop the `next_node_id` synthesis at `:113`.\n- Drop the live-vs-store fork at `:50–78`; the projection is updated as events are written, so a single `state.store.open_run_reader(...).state()` read suffices.\n- Delete `active_stage_state_from_events` at `:19` — no longer needed; `state` is on the projection.\n\n### 5. OpenAPI: extend three schemas\n\nFile: `docs/public/api-reference/fabro-api.yaml`\n\n- **`RunBillingStage`** (`:6610`): add optional `started_at: string (date-time)` and `state: $ref StageState`. Frontend uses `state` to detect in-flight rows.\n- **`RunStage`** (`:6316`): add optional `started_at: string (date-time)`. `status: StageState` already exists.\n- **`StageProjection`** (`:5279`): add optional `started_at`, `duration_ms`, and `state: StageState`. **Do not** add `usage` here — the field is `#[serde(skip)]` server-internal (see §1). `BilledModelUsage` is not currently an OpenAPI schema and modeling it would balloon this PR's surface; `/runs/{id}/state` consumers needing per-stage tokens hit `/billing` instead.\n\nAfter editing: `cargo build -p fabro-api` regenerates Rust types; `cd lib/packages/fabro-api-client && bun run generate` regenerates the TS client.\n\n### 6. Update demo fixtures\n\nFile: `lib/crates/fabro-server/src/demo/mod.rs`\n\n- `RunStage` literals at `:1184, 1191, 1198, 1205` — add `started_at: None`.\n- `RunBillingStage` literals at `:1233, 1252, 1271, 1290` — add `started_at: None` and `state: StageState::Succeeded` (or appropriate per fixture).\n- Any `StageProjection` literals in tests/fixtures — search `rg \"StageProjection \\{\"` and add the new optional fields (typically `..Default::default()` shape if used).\n\n### 7. Frontend: invalidate on stage events + live tick\n\nFiles: `apps/fabro-web/app/lib/run-events.ts`, `apps/fabro-web/app/routes/run-billing.tsx`.\n\n`run-events.ts`:\n- Add `\"stage.retrying\"` to the `STAGE_EVENTS` set at `:35`. The projection now stores Retrying state, so the UI must refetch when this event arrives.\n- Add `queryKeys.runs.billing(runId)` to the `STAGE_EVENTS` invalidation list at `:75`.\n- Update the `queryKeysForRunEvent` test in `run-events.test.tsx` to verify `stage.retrying` invalidates stages, billing, events, and (when stage_id present) stage turns.\n\n`run-billing.tsx`:\n- Detect in-flight via the new `state` field: `state === \"running\" || state === \"retrying\"`.\n- If any row is in-flight, run a `useEffect` `setInterval(..., 1000)` that bumps a `now` state. Render the in-flight row's runtime as `(now − new Date(started_at)) / 1000`.\n- **Footer total**: while ticking, derive total from the rendered row runtimes — sum up the displayed seconds (which now include the live elapsed for the in-flight row). Otherwise (terminal run) use `billing.totals.runtime_secs` from the server.\n- Drop the empty-state at `:83` when any in-flight row exists; the table appears as soon as the first stage starts.\n\nUpdate `apps/fabro-web/app/routes/run-billing.test.tsx`:\n- Extend fixtures with `started_at` and `state`.\n- Add a test for an in-flight row (state = `running`) that asserts (a) the row renders, (b) the footer total includes the elapsed time, (c) the table is shown even when no stage has completed.\n\n### 8. What stays out of scope\n\n- **Live tokens during a stage.** Requires a new `agent.turn.completed { usage }` event from `fabro-agent`/`fabro-llm` plus a reducer arm to accumulate onto `StageProjection.usage`. The schema in §1 is ready; instrumenting it is a separate change.\n- **Per-visit billing rows.** Today's behavior aggregates by node_id (latest visit). One row per retry/revisit is a UX decision separate from this fix.\n- **Removing `checkpoint.node_outcomes`.** Still used by workflow execution: `artifact.rs:92,134`, `finalize.rs:119,394`, retro/conditionals. Leave it.\n- **Mixed in-memory/projection reads on `/checkpoint` and `/graph`.** Different shape of issue; not this PR.\n\n### 9. API round-trip tests\n\nFiles: `lib/crates/fabro-api/tests/stage_projection_round_trip.rs`, `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs`.\n\nExtend the representative-JSON cases:\n\n- `stage_projection_round_trip.rs`: add `started_at`, `duration_ms`, `state` to the JSON fixture and assert they round-trip. Confirms the OpenAPI schema and Rust type stay in lock-step for the new fields.\n- `run_billing_stage_round_trip.rs`: add `started_at` and `state` to the JSON fixture and assert they round-trip. Add a second case for an in-flight row (`state = \"running\"`, no `model`, zero `billing`).\n\nThese prevent silent drift if the OpenAPI schema and Rust type ever diverge on the new fields.\n\n## Files to modify\n\n- `lib/crates/fabro-types/src/run_projection.rs` — fields + helpers\n- `lib/crates/fabro-store/src/run_state.rs` — reducer arms (incl. new `StageRetrying`) + tests\n- `lib/crates/fabro-server/src/server/handler/billing.rs` — both handlers rewritten; delete `active_stage_state_from_events`\n- `lib/crates/fabro-server/src/server/tests.rs` — keep `list_run_stages_projects_retrying_until_completion`; verify it still passes via the new projection-based path\n- `lib/crates/fabro-server/src/demo/mod.rs` — fixture updates\n- `docs/public/api-reference/fabro-api.yaml` — `RunBillingStage`, `RunStage`, `StageProjection`\n- `lib/packages/fabro-api-client` — regenerated\n- `apps/fabro-web/app/lib/run-events.ts` — billing invalidation on stage events\n- `apps/fabro-web/app/routes/run-billing.tsx` — in-flight detection + tick + derived footer total\n- `apps/fabro-web/app/routes/run-billing.test.tsx` — new fixtures + in-flight + footer-tick assertions\n- `lib/crates/fabro-api/tests/stage_projection_round_trip.rs` — extend fixture with new fields\n- `lib/crates/fabro-api/tests/run_billing_stage_round_trip.rs` — extend fixture with new fields, add in-flight case\n- `apps/fabro-web/app/lib/run-events.test.tsx` — assert `stage.retrying` invalidates billing/stages/events\n\n## Existing utilities to reuse\n\n- `RunProjection::iter_stages()` — `lib/crates/fabro-types/src/run_projection.rs:102`\n- `StageProjection::first_event_seq` — already a `NonZeroU32`, ready as sort key\n- `StageState` — `lib/crates/fabro-types/src/outcome.rs:111` with `From` already wired\n- `BilledTokenCounts::from_billed_usage` — used by current totals path\n- `accumulate_model_billing` — `lib/crates/fabro-server/src/server.rs:539`, used for by-model breakdown\n- chrono pattern: `now.signed_duration_since(...).num_milliseconds().max(0) as f64 / 1000.0` (e.g. `lib/crates/fabro-cli/src/commands/runs/list.rs:99`)\n\n## Verification\n\n1. **Reducer unit tests** in `run_state.rs`:\n - `stage_started_records_started_at_and_running_state`\n - `stage_completed_records_duration_usage_and_terminal_state`\n - `stage_failed_records_duration_and_failed_state`\n - `stage_retrying_sets_retrying_state`\n - `stage_started_after_retrying_returns_to_running` (transition)\n2. **Existing test must still pass**: `list_run_stages_projects_retrying_until_completion` (`server/tests.rs:2126`) — covers Retrying via the new projection path.\n3. **New handler integration tests** in `lib/crates/fabro-server/tests/it/scenario/usage.rs`:\n - **Mid-run snapshot**: pause workflow with one completed and one in-flight stage; assert `/billing` returns two rows; in-flight row has `state = \"running\"`, `model = null`, zero `billing` tokens, non-zero `runtime_secs`; totals include the in-flight runtime.\n - **Retried node, mid-retry**: StageStarted → StageFailed (duration_ms = 10) → StageRetrying → StageStarted (no completion yet); assert the row's `state = \"running\"` and `runtime_secs` reflects elapsed since the **second** StageStarted, not the failed attempt's 10ms. Pin the regression risk that motivated the `runtime_secs()` priority inversion.\n - **Retried node, succeeded**: same prefix → StageCompleted; assert one row per node_id (latest visit), state `Succeeded`, duration = final attempt's `duration_ms`.\n - **Revisited node (loop, multi-node)**: emit A completed → B completed → A revisited+completed (visit=2). Assert (a) two rows total, (b) order is A, B (matches `finalize.rs:113`), (c) A's row carries the latest visit's data (visit=2 duration/usage), not the first visit's. Pins both the dedupe rule and the ordering rule against future drift.\n4. **Frontend tests** — `run-billing.test.tsx`:\n - In-flight row renders with runtime > 0.\n - Footer total ticks while the in-flight row ticks.\n - Empty-state hidden when an in-flight row exists.\n5. **End-to-end smoke** — `fabro run repl`, open `/runs//billing` in dev:\n - In-flight stage row appears immediately on `stage.started`.\n - Runtime ticks once per second.\n - On `stage.completed`, row gets `duration_ms` + tokens; next stage's row appears.\n - Footer reflects live in-flight runtime.\n6. **Conformance** — `cargo nextest run -p fabro-server`, `cd apps/fabro-web && bun run typecheck && bun test`, `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`. Run `cargo insta pending-snapshots` afterwards in case any snapshot tests pick up the new optional fields.\n\n## Unresolved questions\n\n- For runs with retried/revisited nodes, is \"latest visit per node_id\" the right billing display, or should we eventually expose all visits as separate rows? Plan matches current behavior; flagging for future.\n- `StageProjection.usage` is server-internal (`#[serde(skip)]`) for this PR. If a future consumer of `/runs/{id}/state` needs per-stage tokens, we'd model `BilledModelUsage` as an OpenAPI schema and unskip it — separate change.\n", + "command.stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "thread.start.current_node": "toolchain", + "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c", + "current_node": "toolchain", + "internal.node_visit_count": 1, + "internal.thread_id": "start", + "internal.work_dir": "/home/daytona/workspace", + "internal.retry_count.start": 0, + "internal.fidelity": "compact" + }, + "node_outcomes": { + "start": { + "status": "succeeded", + "usage": null + }, + "toolchain": { + "status": "succeeded", + "context_updates": { + "command.stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c" + }, + "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", + "usage": null + } + }, + "next_node_id": "preflight_compile", + "git_commit_sha": "a9090e9f2fbcdc142cd959e0e8e897f74ed806de", + "node_visits": { + "toolchain": 1, + "start": 1 + } + } ] ], "conclusion": null, @@ -607,6 +672,23 @@ "superseded_by": null, "pending_interviews": {}, "stages": { + "preflight_compile@1": { + "first_event_seq": 30, + "prompt": null, + "response": null, + "completion": null, + "provider_used": null, + "diff": null, + "script_invocation": { + "script": "cargo check -q --workspace 2>&1", + "command": "cargo check -q --workspace 2>&1", + "language": "shell" + }, + "script_timing": null, + "parallel_results": null, + "stdout": null, + "stderr": null + }, "start@1": { "first_event_seq": 16, "prompt": null, @@ -629,7 +711,12 @@ "first_event_seq": 20, "prompt": null, "response": null, - "completion": null, + "completion": { + "outcome": "succeeded", + "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", + "failure_reason": null, + "timestamp": "2026-05-04T20:07:42.231372Z" + }, "provider_used": null, "diff": null, "script_invocation": { @@ -637,10 +724,25 @@ "command": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", "language": "shell" }, - "script_timing": null, + "script_timing": { + "stdout": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c", + "stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "exit_code": 0, + "duration_ms": 1417, + "termination": "exited", + "stdout_bytes": 36, + "stderr_bytes": 0, + "streams_separated": true, + "live_streaming": true + }, "parallel_results": null, "stdout": null, - "stderr": null + "stderr": null, + "stdout_bytes": 36, + "stderr_bytes": 0, + "streams_separated": true, + "live_streaming": true, + "termination": "exited" } } } \ No newline at end of file diff --git a/stages/002-toolchain@1/script_timing.json b/stages/002-toolchain@1/script_timing.json new file mode 100644 index 000000000..0d1f27622 --- /dev/null +++ b/stages/002-toolchain@1/script_timing.json @@ -0,0 +1,11 @@ +{ + "stdout": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c", + "stderr": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "exit_code": 0, + "duration_ms": 1417, + "termination": "exited", + "stdout_bytes": 36, + "stderr_bytes": 0, + "streams_separated": true, + "live_streaming": true +} \ No newline at end of file diff --git a/stages/002-toolchain@1/status.json b/stages/002-toolchain@1/status.json new file mode 100644 index 000000000..298702308 --- /dev/null +++ b/stages/002-toolchain@1/status.json @@ -0,0 +1,6 @@ +{ + "outcome": "succeeded", + "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", + "failure_reason": null, + "timestamp": "2026-05-04T20:07:42.231372Z" +} \ No newline at end of file diff --git a/stages/002-toolchain@1/stderr.log b/stages/002-toolchain@1/stderr.log new file mode 100644 index 000000000..d87ba9545 --- /dev/null +++ b/stages/002-toolchain@1/stderr.log @@ -0,0 +1 @@ +blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126 \ No newline at end of file diff --git a/stages/002-toolchain@1/stdout.log b/stages/002-toolchain@1/stdout.log new file mode 100644 index 000000000..4e86d161d --- /dev/null +++ b/stages/002-toolchain@1/stdout.log @@ -0,0 +1 @@ +blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c \ No newline at end of file diff --git a/stages/003-preflight_compile@1/script_invocation.json b/stages/003-preflight_compile@1/script_invocation.json new file mode 100644 index 000000000..16acaaf06 --- /dev/null +++ b/stages/003-preflight_compile@1/script_invocation.json @@ -0,0 +1,5 @@ +{ + "script": "cargo check -q --workspace 2>&1", + "command": "cargo check -q --workspace 2>&1", + "language": "shell" +} \ No newline at end of file