From 05d1dc5777481dc84ebdf0298f2b94f7241c3568 Mon Sep 17 00:00:00 2001 From: Fabro Date: Mon, 25 May 2026 19:01:20 -0400 Subject: [PATCH] =?UTF-8?q?finalize=20run=20=E2=9A=92=EF=B8=8F=20Generated?= =?UTF-8?q?=20with=20[Fabro](https://fabro.sh)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- run.json | 345 +++++++++++++++++++------ stages/008-verify@1/diff.patch | 244 +++++++++++++++++ stages/008-verify@1/output.log | 1 + stages/008-verify@1/script_timing.json | 8 + stages/008-verify@1/status.json | 6 + stages/009-exit@1/status.json | 6 + 6 files changed, 526 insertions(+), 84 deletions(-) create mode 100644 stages/008-verify@1/diff.patch create mode 100644 stages/008-verify@1/output.log create mode 100644 stages/008-verify@1/script_timing.json create mode 100644 stages/008-verify@1/status.json create mode 100644 stages/009-exit@1/status.json diff --git a/run.json b/run.json index fefe3e3eb..d1996686f 100644 --- a/run.json +++ b/run.json @@ -505,7 +505,7 @@ "kind": "running" }, "status_updated_at": "2026-05-25T21:44:41.751851Z", - "last_event_at": "2026-05-25T22:52:16.027281Z", + "last_event_at": "2026-05-25T23:01:20.422027Z", "pending_control": null, "checkpoints": [ { @@ -1247,9 +1247,9 @@ } }, { - "seq": 0, + "seq": 1829, "checkpoint": { - "timestamp": "2026-05-25T23:01:16.017079Z", + "timestamp": "2026-05-25T23:01:20.421901Z", "current_node": "verify", "completed_nodes": [ "start", @@ -1263,55 +1263,85 @@ ], "node_retries": {}, "context_values": { - "graph.rankdir": "LR", - "failure_class": "", - "response.implement": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - Exposes `SessionInputTiming` via `Session::last_input_timing()`.\n - Resets timing at each `process_input_with_runtime` entry.\n- Threaded timing through workflow LLM backends:\n - `CodergenResult::Text` now carries active `StageTiming`.\n - API backend measures one-shot inference and agent-loop inference/tool time.\n - ACP backend attributes opaque duration to inference.\n- Populated `Outcome.timing` in production handlers:\n - Agent, prompt, fan-in, and command handlers now preserve active timing.\n - Command duration is reported as tool time.\n- Preserved timing/billing for terminal run failures:\n - `persist_terminal_engine_failure` now uses `build_conclusion_from_store` timing and billing.\n - Detached bootstrap/completion guards now preserve conclusion timing/billing when a store exists.\n - Detached failure helper now rolls up timing/billing from the store.\n- Added tests for:\n - Agent per-input inference/tool timing.\n - Agent/prompt/fan-in handler timing propagation.\n - Command handler tool timing.\n - Executor propagation into `NodeResult`.\n - Full workflow event rollup from `Outcome.timing` → `stage.completed` → `run.completed`.\n - Terminal engine failure and drop-guard timing/billing preservation.\n\nValidation passed:\n- `cargo nextest run -p fabro-agent -p fabro-workflow -p fabro-store -p fabro-core`\n - 1913 passed, 122 skipped\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`\n- `cargo build -p fabro-server`\n- `git diff --check`\n\nI did not run the interactive/live server + UI E2E workflow steps, but the server build and full requested automated validation are green.", - "internal.retry_count.start": 0, - "thread.simplify_opus.current_node": "simplify_gpt", - "last_stage": "simplify_gpt", - "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", - "internal.work_dir": "/home/daytona/workspace/fabro", - "outcome": "succeeded", "command.output": "blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d", - "internal.retry_count.toolchain": 0, - "thread.preflight_compile.current_node": "preflight_lint", - "response.simplify_gpt": "Reviewed the branch diff vs merge base and ran the parallel review pass. I found and fixed one cleanup issue:\n\n- Restored `Session::process_input_with_runtime` to its original public `Result<(), Error>` shape instead of returning a tuple.\n- Added the requested `Session::last_input_timing()` getter and `last_input_timing` state on `Session`.\n- Updated workflow callers to read timing through the getter after `process_input_with_runtime(...)`.\n- Updated the agent timing unit test accordingly.\n\nValidation passed:\n\n- `cargo check -q -p fabro-agent -p fabro-workflow`\n- `cargo nextest run -p fabro-agent -p fabro-workflow`\n- `cargo nextest run -p fabro-store -p fabro-core`\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy -q -p fabro-agent -p fabro-workflow --all-targets -- -D warnings`\n\nWorking tree now has the cleanup edits in:\n\n- `lib/crates/fabro-agent/src/session.rs`\n- `lib/crates/fabro-workflow/src/handler/llm/api.rs`", - "thread.start.current_node": "toolchain", + "failure_class": "", + "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", "internal.retry_count.preflight_compile": 0, - "internal.retry_count.simplify_gpt": 0, - "internal.fidelity": "compact", - "current_node": "verify", - "internal.retry_count.simplify_opus": 0, + "internal.retry_count.start": 0, + "response.simplify_opus": "All six fixes landed, with net **-57 lines** even while adding three shared helpers and a constructor. Summary of cleanups applied based on the three reviewers' findings:\n\n## Fixed\n\n1. **Returned timing from `process_input_with_runtime` instead of stashing it on `Session`** (Agent 2 #1). Dropped `last_input_inference_duration` / `last_input_tool_duration` fields, the `last_input_timing()` getter, and the per-call reset prologue. Signature is now `-> (SessionInputTiming, Result<(), Error>)`. Eliminates the \"must read immediately\" implicit contract.\n\n2. **Replaced the `record_inference_duration!` macro with a free function** `record_elapsed(&mut Option, &mut Duration)` (Agent 1 #2, Agent 2 #2). Same 7 call sites, but type-checked, IDE-discoverable, and no macro hygiene.\n\n3. **Made `run_store` required on `DetachedRunBootstrapGuard::arm()`** (Agent 1 #7, Agent 2 #3/#5). Dropped the `Option` field and the external setter pattern. Drop impl unconditionally builds the conclusion now.\n\n4. **Extracted `emit_workflow_run_failed` helper** (Agent 1 #7, Agent 2 #4). Collapses the four near-identical \"build conclusion → emit run.failed\" sites in `start.rs` (terminal engine failure, both Drop guards, `persist_detached_failure`) into one shared function.\n\n5. **Added `StageTiming::active_only(inference, tool)` constructor** (Agent 1 #1). Removed the local `active_stage_timing` helper in `handler/llm/api.rs` and unified `acp.rs` and `command.rs` on the same shape.\n\n6. **Reverted `Box` in `CodergenResult::Text`** (Agent 1 #6, Agent 2 #6). Was the worst of both worlds — didn't equalise variant size, forced `.map(|usage| *usage)` ceremony at every call site. Used `#[allow(clippy::large_enum_variant, reason = \"...\")]` matching the established pattern in `event/events.rs` and `event/sink.rs`.\n\n7. **Hoisted `test_usage` and `mark_run_running` to `test_support`** (Agent 1 #3/#4). `test_usage` was duplicated in 4 files (this PR was about to add a 5th); the 4th `mark_run_running` copy was right next to a pre-existing 3-event seed in test_support that didn't quite match. Both are now public helpers in `crate::test_support`.\n\n## Skipped (with reason)\n\n- **Make Drop guard tasks survive runtime shutdown** (Agent 3 #3) — flagged as a correctness-on-shutdown concern, not introduced by this PR; widening it would be a separate fix.\n- **`wait_for_conclusion` polling timeout hardening** (Agent 2 #8) — minor CI flake risk; current 50ms budget is enough for unit tests.\n- **`Candidate::timing` zero on LLM-failure heuristic fallback** (Agent 2 #9, Agent 3 implicit) — agent itself noted this is acceptable; capturing wasted inference on errors would require threading Instants through the fan-in evaluator.\n- **Helper for the 3-line failed-outcome decoration in agent.rs/prompt.rs** (Agent 2 #7/#10) — marginal benefit not worth the indirection.\n\n## Verification\n\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` clean\n- `cargo +nightly-2026-04-14 fmt --check --all` clean\n- `cargo nextest run -p fabro-workflow -p fabro-agent -p fabro-types -p fabro-core` — all 3592 tests pass (1175 + 1844 + 573 across the affected crates).", + "last_stage": "simplify_gpt", + "internal.retry_count.preflight_lint": 0, + "internal.retry_count.verify": 0, "last_response": "Reviewed the branch diff vs merge base and ran the parallel review pass. I found and fixed one cleanup issue:\n\n- Restored `Session::process_input_with_runtime` to its original public `Result<(), Error", + "outcome": "succeeded", + "internal.retry_count.simplify_gpt": 0, + "response.simplify_gpt": "Reviewed the branch diff vs merge base and ran the parallel review pass. I found and fixed one cleanup issue:\n\n- Restored `Session::process_input_with_runtime` to its original public `Result<(), Error>` shape instead of returning a tuple.\n- Added the requested `Session::last_input_timing()` getter and `last_input_timing` state on `Session`.\n- Updated workflow callers to read timing through the getter after `process_input_with_runtime(...)`.\n- Updated the agent timing unit test accordingly.\n\nValidation passed:\n\n- `cargo check -q -p fabro-agent -p fabro-workflow`\n- `cargo nextest run -p fabro-agent -p fabro-workflow`\n- `cargo nextest run -p fabro-store -p fabro-core`\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy -q -p fabro-agent -p fabro-workflow --all-targets -- -D warnings`\n\nWorking tree now has the cleanup edits in:\n\n- `lib/crates/fabro-agent/src/session.rs`\n- `lib/crates/fabro-workflow/src/handler/llm/api.rs`", + "thread.preflight_compile.current_node": "preflight_lint", + "thread.preflight_lint.current_node": "implement", + "failure_signature": "", + "thread.simplify_gpt.current_node": "verify", + "thread.simplify_opus.current_node": "simplify_gpt", "thread.toolchain.current_node": "preflight_compile", - "internal.node_visit_count": 1, "internal.thread_id": "simplify_gpt", "graph.goal": "# Plan: Fix stage timing (inference + tool) reporting\n\n## Context\n\nThe web UI's Duration popover shows `Active (inference + tools): 0ms` for every run, including agent-heavy runs that obviously did substantial LLM and tool work. Verified on `01KSE2PAVXD56N4TWNK4T5H5VA`: 10 stage.completed events and 1 run.failed event all carry `inference_time_ms: 0, tool_time_ms: 0`, even though stages like `implement@1` (94 min wall) and `simplify_opus@1` (29 min wall) were doing nothing but inference and tool calls.\n\nTwo independent bugs:\n\n1. **No production handler ever populates `Outcome.timing`.** The plumbing from `Outcome.timing` → `NodeResult` (`lib/crates/fabro-core/src/executor.rs:30-37`) → `StageTiming` → `stage.completed` props → projection → billing rollup → run.completed/failed → UI is fully wired and shipped as of #343 (2026-05-21), but `AgentHandler::execute`, `PromptHandler::execute`, `CommandHandler::execute`, and `FanInHandler` all build `Outcome::success()` and never touch `.timing`. The executor falls back to zero, and every downstream consumer faithfully aggregates zero.\n\n2. **`persist_terminal_engine_failure` and its sibling Drop-guard failure paths discard timing/billing entirely.** When the engine returns `Err` (e.g. `VisitLimitExceeded`, which is what killed the user's run), `lib/crates/fabro-workflow/src/operations/start.rs:284-308` builds a `Conclusion` via `build_conclusion_from_store`, then throws it away (`let _conclusion = ...`) and emits `WorkflowRunFailed` with `RunTiming::wall_only(...)` and `None` for billing/diff. The three Drop-guard paths (`start.rs:934`, `1001`, `1033`) do similar with `RunTiming::default()` and never even build a conclusion.\n\nGoal: stage and run events carry real per-stage `inference_time_ms` + `tool_time_ms`; engine-failure terminal events preserve the conclusion's rolled-up timing and billing.\n\n## Approach\n\n### Part A — Capture inference + tool time in handlers (Bug 1)\n\n**A1. `fabro-agent` — accumulate per-input timing in `Session`**\n\n`lib/crates/fabro-agent/src/session.rs`\n\nAdd two `Duration` accumulators to `Session` (initialised to `Duration::ZERO`):\n- `last_input_inference_duration`\n- `last_input_tool_duration`\n\nIn `process_input_with_runtime` (line 1196), zero them at entry so each call's totals are independent.\n\nIn `run_single_input` (line 1254):\n- Wrap the inference span: capture `Instant::now()` immediately before opening the stream at line 1391, and add `.elapsed()` to `last_input_inference_duration` once `response = Some(resp)` (line 1487-1490) OR when the loop exits with an error/cancellation. The whole `'streamattempts` loop counts as inference work — retries included.\n- Wrap the tool span around `execute_tool_calls` at line 1705-1719: `Instant::now()` before, accumulate `.elapsed()` after `.await`.\n\nExpose a getter:\n```rust\npub fn last_input_timing(&self) -> SessionInputTiming { ... }\n```\nwhere `SessionInputTiming { pub inference: Duration, pub tool: Duration }` is a new tiny struct in `fabro-agent`.\n\n**A2. `fabro-workflow` — thread timing through the backend boundary**\n\n`lib/crates/fabro-workflow/src/handler/agent.rs`\n\nExtend `CodergenResult::Text` with a `timing: fabro_types::StageTiming` field (wall is irrelevant — see note below). Update the few `CodergenResult::Text { ... }` constructions found by the explore agent to populate it; existing match-bindings only read `text`/`usage`/`files_touched` so they keep compiling with `..` patterns. `CodergenResult::Full(outcome)` keeps current behaviour — the outcome itself already carries any timing.\n\nNote on wall: `lib/crates/fabro-core/src/executor.rs:30-37` reads ONLY `inference_time_ms` and `tool_time_ms` out of `outcome.timing`. The wall comes from the executor's own stopwatch. So we construct `StageTiming::new(0, inference_ms, tool_ms)` and document that the wall field is ignored in this hop.\n\n`lib/crates/fabro-workflow/src/handler/llm/api.rs`\n\n- `AgentApiBackend::run` (line 1103): after `session.process_input_with_runtime(...)` returns, read `session.last_input_timing()` and set the new `timing` on `CodergenResult::Text` at line 1094.\n- `AgentApiBackend::one_shot` (line 994): wrap the `complete_one_shot_request` call at line 1053 with `Instant::now()` / `.elapsed()`. Accumulate across repair iterations of the surrounding loop. All of it counts as inference; no tool work happens in `one_shot`. Set `timing` on `CodergenResult::Text` at line 1094.\n\n`lib/crates/fabro-workflow/src/handler/llm/acp.rs`\n\n`AgentAcpBackend::run` (line ~140): already exposes `result.duration_ms`. Set `timing: StageTiming::new(0, duration_ms, 0)` on the returned `CodergenResult::Text` (per user decision: attribute all ACP duration to inference; ACP is opaque about the split).\n\n**A3. Consume timing in stage handlers and set `outcome.timing`**\n\n- `lib/crates/fabro-workflow/src/handler/agent.rs:341` — after building `outcome`, before the final `Ok(outcome)`, set `outcome.timing = Some(timing_from_codergen_result)`.\n- `lib/crates/fabro-workflow/src/handler/prompt.rs:180` — same pattern.\n- `lib/crates/fabro-workflow/src/handler/fan_in.rs:266` — backend returns timing; pass it onto the outcome built from the fan-in response.\n- `lib/crates/fabro-workflow/src/handler/command.rs:175` — `outcome.timing = Some(StageTiming::new(0, 0, result.duration_ms))`. All command wall-time is tool time. `result.duration_ms` is already at line 154 in scope.\n\nOther handlers (`human`, `wait`, `conditional`, `parallel`, `start`, `exit`, `structured_output`, `manager_loop`) do no inference or tool work. Leave `outcome.timing` as `None`; the executor will naturally produce `inference: 0, tool: 0` for those stages, which is correct.\n\n### Part B — Preserve conclusion timing on engine failure (Bug 2)\n\n`lib/crates/fabro-workflow/src/operations/start.rs`\n\n**B1. Main path** (`persist_terminal_engine_failure`, line 274-308):\n- Rename `_conclusion` → `conclusion` and use it:\n - Pass `conclusion.timing` (already a `RunTiming` with the proper inference/tool/wall rollup from `build_conclusion_from_parts`) instead of `RunTiming::wall_only(...)`.\n - Pass `conclusion.billing.clone()` instead of `None` for the billing arg of `workflow_run_failed_from_error`.\n - `final_git_commit_sha`, `final_patch`, `diff_summary` stay `None` — those require the finalize-side workspace diff computation that this path deliberately skips.\n\n**B2. Drop-guard paths** (per user decision: fix them too):\n\n- `DetachedRunBootstrapGuard` (line 882-948): add an `Option` field. The bootstrap function builds the guard before the store exists, then mutates `bootstrap_guard.run_store = Some(store.clone())` once the store is in scope. On Drop, if the store is `Some`, the spawned task calls `build_conclusion_from_store` and uses its timing/billing; otherwise falls back to `RunTiming::default()` (pre-store failure means no stages can possibly exist).\n\n- `DetachedRunCompletionGuard` (line 953-1021): armed after the store exists, so add a non-optional `run_store: RunStoreHandle`. Drop's spawned task builds the conclusion and uses it.\n\n- `persist_detached_failure` (line 1023): add a `run_store: &RunStoreHandle` parameter. Call `build_conclusion_from_store` and forward `timing` + `billing` to the failure event. Update the two callers (postrun-related) to pass the store they already have in scope.\n\n`RunStoreHandle` is already `Clone` (the surrounding code clones it routinely), so move-into-spawned-task is fine.\n\n### Critical existing utilities to reuse (do not duplicate)\n\n- `fabro_types::StageTiming::new(wall, inference, tool)` and `RunTiming::new(...)` — invariant-enforcing constructors at `lib/crates/fabro-types/src/timing.rs:38, 91`.\n- `crate::millis_u64(duration)` helper for `Duration → u64` ms in `fabro-workflow` (used widely; see `lifecycle/event.rs:80-86`).\n- `build_conclusion_from_store` at `lib/crates/fabro-workflow/src/pipeline/finalize.rs:71` already does the rollup we need on the engine-failure path.\n- `billing_rollup_from_projection` (called inside `build_conclusion_from_parts`) sums per-stage timings into `RunTiming` — no need to reimplement.\n\n## Tests\n\n- **`fabro-agent` unit test**: feed `Session` a fake `LlmClient` whose `stream` sleeps a known duration and a fake tool that sleeps another known duration. Drive one `process_input_with_runtime` call. Assert `session.last_input_timing()` reports both non-zero and roughly matching the sleeps. Then call again and assert it's per-call (not cumulative).\n- **`fabro-workflow` handler tests**: in `handler/agent.rs`'s test module, wire a `CodergenBackend` that returns `CodergenResult::Text { timing: StageTiming::new(0, 200, 300), .. }` and assert `AgentHandler::execute`'s returned `Outcome.timing` carries those values. Mirror for `prompt.rs` and `fan_in.rs`. Add a `command.rs` test that mocks a `sandbox.exec_command_streaming` returning `duration_ms = 500` and asserts `outcome.timing.tool_time_ms == 500`.\n- **Executor integration**: add a test in `fabro-workflow` (or extend an existing one in `pipeline/finalize.rs` tests) that runs a tiny graph with a handler producing `Outcome.timing = Some(StageTiming::new(0, 100, 50))` and asserts the emitted `stage.completed` event carries those values, and that `run.completed` carries the summed rollup.\n- **`persist_terminal_engine_failure` test**: seed a `RunStore` with a couple of `stage.completed` events whose timing is non-zero, drive the engine-failure path, and assert the emitted `WorkflowRunFailed` event has `timing.inference_time_ms` and `tool_time_ms` matching the per-stage sum and `billing` populated.\n- **Drop guard tests**: trickier because of `Handle::try_current` + spawn. Add focused tests that arm a guard, drop it, and `tokio::task::yield_now().await` enough times to let the spawned task run, then assert the emitted failure event carries non-zero timing.\n- Run `cargo nextest run -p fabro-agent -p fabro-workflow -p fabro-store -p fabro-core`.\n- Run formatter and lints per CLAUDE.md: `cargo +nightly-2026-04-14 fmt --check --all` and `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`.\n\n## End-to-end verification\n\n1. Build the server: `cargo build -p fabro-server`.\n2. Start server: `fabro server start`.\n3. Run a small agent-backed workflow (e.g. `fabro run repl` with a short prompt that fires at least one tool call).\n4. `fabro events --json | jq -s '[.[] | select(.event==\"stage.completed\")] | .[].properties.timing'` — confirm `inference_time_ms > 0` and `tool_time_ms > 0` for the agent stage.\n5. `fabro events --json | jq -s '[.[] | select(.event==\"run.completed\" or .event==\"run.failed\")] | .[].properties.timing'` — confirm `active_time_ms == inference_time_ms + tool_time_ms` and both are non-zero.\n6. Open the run in the web UI (start the SPA dev build per CLAUDE.md or rebuild the embedded SPA with `cargo dev build`), hover the Duration chip, confirm **Active (inference + tools)** is non-zero.\n7. For Bug 2: force an engine failure by setting a very low visit limit and rerunning the same workflow; confirm the `run.failed` event timing breakdown is non-zero and matches the per-stage sum.\n\n## Out of scope\n\n- Adding `wall_time_ms` correctness to `Outcome.timing` (executor ignores it; doc tweak only if necessary).\n- Surfacing inference vs tool split for ACP backend beyond \"all-inference\" attribution.\n- Backfilling timing for historical runs that have already emitted zero events — past events are immutable.\n- Web UI changes beyond what the existing popover already renders.\n", - "thread.simplify_gpt.current_node": "verify", - "failure_signature": "", - "response.simplify_opus": "All six fixes landed, with net **-57 lines** even while adding three shared helpers and a constructor. Summary of cleanups applied based on the three reviewers' findings:\n\n## Fixed\n\n1. **Returned timing from `process_input_with_runtime` instead of stashing it on `Session`** (Agent 2 #1). Dropped `last_input_inference_duration` / `last_input_tool_duration` fields, the `last_input_timing()` getter, and the per-call reset prologue. Signature is now `-> (SessionInputTiming, Result<(), Error>)`. Eliminates the \"must read immediately\" implicit contract.\n\n2. **Replaced the `record_inference_duration!` macro with a free function** `record_elapsed(&mut Option, &mut Duration)` (Agent 1 #2, Agent 2 #2). Same 7 call sites, but type-checked, IDE-discoverable, and no macro hygiene.\n\n3. **Made `run_store` required on `DetachedRunBootstrapGuard::arm()`** (Agent 1 #7, Agent 2 #3/#5). Dropped the `Option` field and the external setter pattern. Drop impl unconditionally builds the conclusion now.\n\n4. **Extracted `emit_workflow_run_failed` helper** (Agent 1 #7, Agent 2 #4). Collapses the four near-identical \"build conclusion → emit run.failed\" sites in `start.rs` (terminal engine failure, both Drop guards, `persist_detached_failure`) into one shared function.\n\n5. **Added `StageTiming::active_only(inference, tool)` constructor** (Agent 1 #1). Removed the local `active_stage_timing` helper in `handler/llm/api.rs` and unified `acp.rs` and `command.rs` on the same shape.\n\n6. **Reverted `Box` in `CodergenResult::Text`** (Agent 1 #6, Agent 2 #6). Was the worst of both worlds — didn't equalise variant size, forced `.map(|usage| *usage)` ceremony at every call site. Used `#[allow(clippy::large_enum_variant, reason = \"...\")]` matching the established pattern in `event/events.rs` and `event/sink.rs`.\n\n7. **Hoisted `test_usage` and `mark_run_running` to `test_support`** (Agent 1 #3/#4). `test_usage` was duplicated in 4 files (this PR was about to add a 5th); the 4th `mark_run_running` copy was right next to a pre-existing 3-event seed in test_support that didn't quite match. Both are now public helpers in `crate::test_support`.\n\n## Skipped (with reason)\n\n- **Make Drop guard tasks survive runtime shutdown** (Agent 3 #3) — flagged as a correctness-on-shutdown concern, not introduced by this PR; widening it would be a separate fix.\n- **`wait_for_conclusion` polling timeout hardening** (Agent 2 #8) — minor CI flake risk; current 50ms budget is enough for unit tests.\n- **`Candidate::timing` zero on LLM-failure heuristic fallback** (Agent 2 #9, Agent 3 implicit) — agent itself noted this is acceptable; capturing wasted inference on errors would require threading Instants through the fan-in evaluator.\n- **Helper for the 3-line failed-outcome decoration in agent.rs/prompt.rs** (Agent 2 #7/#10) — marginal benefit not worth the indirection.\n\n## Verification\n\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` clean\n- `cargo +nightly-2026-04-14 fmt --check --all` clean\n- `cargo nextest run -p fabro-workflow -p fabro-agent -p fabro-types -p fabro-core` — all 3592 tests pass (1175 + 1844 + 573 across the affected crates).", - "internal.run_id": "01KSGHHBR7DQ1RHFYKD7P46R6F", - "internal.retry_count.preflight_lint": 0, + "internal.retry_count.toolchain": 0, + "internal.work_dir": "/home/daytona/workspace/fabro", + "response.implement": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - Exposes `SessionInputTiming` via `Session::last_input_timing()`.\n - Resets timing at each `process_input_with_runtime` entry.\n- Threaded timing through workflow LLM backends:\n - `CodergenResult::Text` now carries active `StageTiming`.\n - API backend measures one-shot inference and agent-loop inference/tool time.\n - ACP backend attributes opaque duration to inference.\n- Populated `Outcome.timing` in production handlers:\n - Agent, prompt, fan-in, and command handlers now preserve active timing.\n - Command duration is reported as tool time.\n- Preserved timing/billing for terminal run failures:\n - `persist_terminal_engine_failure` now uses `build_conclusion_from_store` timing and billing.\n - Detached bootstrap/completion guards now preserve conclusion timing/billing when a store exists.\n - Detached failure helper now rolls up timing/billing from the store.\n- Added tests for:\n - Agent per-input inference/tool timing.\n - Agent/prompt/fan-in handler timing propagation.\n - Command handler tool timing.\n - Executor propagation into `NodeResult`.\n - Full workflow event rollup from `Outcome.timing` → `stage.completed` → `run.completed`.\n - Terminal engine failure and drop-guard timing/billing preservation.\n\nValidation passed:\n- `cargo nextest run -p fabro-agent -p fabro-workflow -p fabro-store -p fabro-core`\n - 1913 passed, 122 skipped\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`\n- `cargo build -p fabro-server`\n- `git diff --check`\n\nI did not run the interactive/live server + UI E2E workflow steps, but the server build and full requested automated validation are green.", + "thread.implement.current_node": "simplify_opus", + "thread.start.current_node": "toolchain", + "graph.rankdir": "LR", + "internal.node_visit_count": 1, "internal.retry_count.implement": 0, - "thread.preflight_lint.current_node": "implement", - "internal.retry_count.verify": 0, - "thread.implement.current_node": "simplify_opus" + "internal.run_id": "01KSGHHBR7DQ1RHFYKD7P46R6F", + "current_node": "verify", + "internal.fidelity": "compact", + "internal.retry_count.simplify_opus": 0 }, "node_outcomes": { - "toolchain": { + "implement": { "status": "succeeded", "context_updates": { - "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c" + "response.implement": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - Exposes `SessionInputTiming` via `Session::last_input_timing()`.\n - Resets timing at each `process_input_with_runtime` entry.\n- Threaded timing through workflow LLM backends:\n - `CodergenResult::Text` now carries active `StageTiming`.\n - API backend measures one-shot inference and agent-loop inference/tool time.\n - ACP backend attributes opaque duration to inference.\n- Populated `Outcome.timing` in production handlers:\n - Agent, prompt, fan-in, and command handlers now preserve active timing.\n - Command duration is reported as tool time.\n- Preserved timing/billing for terminal run failures:\n - `persist_terminal_engine_failure` now uses `build_conclusion_from_store` timing and billing.\n - Detached bootstrap/completion guards now preserve conclusion timing/billing when a store exists.\n - Detached failure helper now rolls up timing/billing from the store.\n- Added tests for:\n - Agent per-input inference/tool timing.\n - Agent/prompt/fan-in handler timing propagation.\n - Command handler tool timing.\n - Executor propagation into `NodeResult`.\n - Full workflow event rollup from `Outcome.timing` → `stage.completed` → `run.completed`.\n - Terminal engine failure and drop-guard timing/billing preservation.\n\nValidation passed:\n- `cargo nextest run -p fabro-agent -p fabro-workflow -p fabro-store -p fabro-core`\n - 1913 passed, 122 skipped\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`\n- `cargo build -p fabro-server`\n- `git diff --check`\n\nI did not run the interactive/live server + UI E2E workflow steps, but the server build and full requested automated validation are green.", + "last_stage": "implement", + "last_response": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - " }, - "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", + "notes": "Stage completed: implement", + "usage": { + "input": { + "usage": { + "model": { + "provider": "openai", + "model_id": "gpt-5.5" + }, + "tokens": { + "input_tokens": 2811275, + "output_tokens": 9602, + "reasoning_tokens": 9372, + "cache_read_tokens": 6235136, + "cache_write_tokens": 0 + } + }, + "facts": { + "algorithm": "openai" + } + }, + "total_usd_micros": 17743163 + } + }, + "verify": { + "status": "succeeded", + "context_updates": { + "command.output": "blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d" + }, + "notes": "Script completed: git fetch origin main 2>&1 && git merge --no-edit --no-stat origin/main 2>&1 && cargo +nightly-2026-04-14 fmt --all 2>&1 && cargo dev docs refresh 2>&1 && cargo +nightly-2026-04-14 fmt --check --all 2>&1 && { command -v rg >/dev/null 2>&1 || { echo 'rg is required for verify'; exit 127; }; } && ! rg -n 'AuthMode::Disabled|RunAuthMethod|RunSubjectProvenance|\\bActorRef\\b|\\bActorKind\\b|AuthenticatedSubject|AuthenticatedService|AuthorizeRunScoped|AuthorizeRunBlob|AuthorizeStageArtifact|AuthorizeCommandLog|auth_method\\s*==\\s*\"disabled\"' lib/crates apps lib/packages docs/public/api-reference/fabro-api.yaml 2>&1 && cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --workspace --status-level slow --profile ci 2>&1 && cargo dev docs check 2>&1 && bun install --frozen-lockfile 2>&1 && (cd apps/fabro-web && bun run typecheck) 2>&1 && (cd apps/fabro-web && bun run test) 2>&1 && (cd lib/packages/fabro-api-client && bun run typecheck) 2>&1 && cargo dev build -- -p fabro-cli --release 2>&1", "usage": null }, - "preflight_compile": { + "preflight_lint": { "status": "succeeded", "context_updates": { "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" }, - "notes": "Script completed: cargo check -q --workspace 2>&1", + "notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", "usage": null }, "simplify_opus": { @@ -1360,35 +1390,17 @@ "/home/daytona/workspace/fabro/lib/crates/fabro-workflow/src/test_support.rs" ] }, - "implement": { + "start": { + "status": "succeeded", + "usage": null + }, + "preflight_compile": { "status": "succeeded", "context_updates": { - "response.implement": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - Exposes `SessionInputTiming` via `Session::last_input_timing()`.\n - Resets timing at each `process_input_with_runtime` entry.\n- Threaded timing through workflow LLM backends:\n - `CodergenResult::Text` now carries active `StageTiming`.\n - API backend measures one-shot inference and agent-loop inference/tool time.\n - ACP backend attributes opaque duration to inference.\n- Populated `Outcome.timing` in production handlers:\n - Agent, prompt, fan-in, and command handlers now preserve active timing.\n - Command duration is reported as tool time.\n- Preserved timing/billing for terminal run failures:\n - `persist_terminal_engine_failure` now uses `build_conclusion_from_store` timing and billing.\n - Detached bootstrap/completion guards now preserve conclusion timing/billing when a store exists.\n - Detached failure helper now rolls up timing/billing from the store.\n- Added tests for:\n - Agent per-input inference/tool timing.\n - Agent/prompt/fan-in handler timing propagation.\n - Command handler tool timing.\n - Executor propagation into `NodeResult`.\n - Full workflow event rollup from `Outcome.timing` → `stage.completed` → `run.completed`.\n - Terminal engine failure and drop-guard timing/billing preservation.\n\nValidation passed:\n- `cargo nextest run -p fabro-agent -p fabro-workflow -p fabro-store -p fabro-core`\n - 1913 passed, 122 skipped\n- `cargo +nightly-2026-04-14 fmt --check --all`\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`\n- `cargo build -p fabro-server`\n- `git diff --check`\n\nI did not run the interactive/live server + UI E2E workflow steps, but the server build and full requested automated validation are green.", - "last_stage": "implement", - "last_response": "Implemented the stage timing and terminal failure rollup fixes.\n\nSummary of changes:\n- Added per-input `fabro-agent::Session` timing:\n - Tracks LLM inference/stream time and tool execution time.\n - " + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" }, - "notes": "Stage completed: implement", - "usage": { - "input": { - "usage": { - "model": { - "provider": "openai", - "model_id": "gpt-5.5" - }, - "tokens": { - "input_tokens": 2811275, - "output_tokens": 9602, - "reasoning_tokens": 9372, - "cache_read_tokens": 6235136, - "cache_write_tokens": 0 - } - }, - "facts": { - "algorithm": "openai" - } - }, - "total_usd_micros": 17743163 - } + "notes": "Script completed: cargo check -q --workspace 2>&1", + "usage": null }, "simplify_gpt": { "status": "succeeded", @@ -1420,43 +1432,153 @@ "total_usd_micros": 4885902 } }, - "verify": { + "toolchain": { "status": "succeeded", "context_updates": { - "command.output": "blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d" + "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c" }, - "notes": "Script completed: git fetch origin main 2>&1 && git merge --no-edit --no-stat origin/main 2>&1 && cargo +nightly-2026-04-14 fmt --all 2>&1 && cargo dev docs refresh 2>&1 && cargo +nightly-2026-04-14 fmt --check --all 2>&1 && { command -v rg >/dev/null 2>&1 || { echo 'rg is required for verify'; exit 127; }; } && ! rg -n 'AuthMode::Disabled|RunAuthMethod|RunSubjectProvenance|\\bActorRef\\b|\\bActorKind\\b|AuthenticatedSubject|AuthenticatedService|AuthorizeRunScoped|AuthorizeRunBlob|AuthorizeStageArtifact|AuthorizeCommandLog|auth_method\\s*==\\s*\"disabled\"' lib/crates apps lib/packages docs/public/api-reference/fabro-api.yaml 2>&1 && cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --workspace --status-level slow --profile ci 2>&1 && cargo dev docs check 2>&1 && bun install --frozen-lockfile 2>&1 && (cd apps/fabro-web && bun run typecheck) 2>&1 && (cd apps/fabro-web && bun run test) 2>&1 && (cd lib/packages/fabro-api-client && bun run typecheck) 2>&1 && cargo dev build -- -p fabro-cli --release 2>&1", - "usage": null - }, - "preflight_lint": { - "status": "succeeded", - "context_updates": { - "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" - }, - "notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", - "usage": null - }, - "start": { - "status": "succeeded", + "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", "usage": null } }, "next_node_id": "exit", + "git_commit_sha": "f6d3ed18abc5fc51efb23a3aaefbe97919540988", "node_visits": { "verify": 1, - "simplify_opus": 1, "preflight_compile": 1, + "preflight_lint": 1, + "start": 1, + "toolchain": 1, "implement": 1, "simplify_gpt": 1, - "toolchain": 1, - "start": 1, - "preflight_lint": 1 + "simplify_opus": 1 } }, - "diff": {} + "diff": { + "patch": "diff --git a/apps/fabro-web/app/routes/settings-monitoring.test.tsx b/apps/fabro-web/app/routes/settings-monitoring.test.tsx\nindex 10227b98a..070276f8d 100644\n--- a/apps/fabro-web/app/routes/settings-monitoring.test.tsx\n+++ b/apps/fabro-web/app/routes/settings-monitoring.test.tsx\n@@ -106,7 +106,7 @@ function sampleServerSettings(maxConcurrentRuns = 8): ServerSettings {\n describe(\"SettingsMonitoring route\", () => {\n beforeEach(() => {\n teardownReactTestEnv = setupReactTestEnv();\n- systemInfo = { runs: { active: 3, total: 12 } };\n+ systemInfo = { runs: { active: 3, scheduler_slots_used: 1, total: 12 } };\n serverSettings = sampleServerSettings();\n });\n \n@@ -133,7 +133,8 @@ describe(\"SettingsMonitoring route\", () => {\n expect(text).toContain(\"5s\");\n expect(text).toContain(\"3 GiB\");\n expect(text).toContain(\"8 GiB\");\n- expect(text).toContain(\"3 / 8 active\");\n+ expect(text).toContain(\"1 / 8 slots used\");\n+ expect(text).not.toContain(\"3 / 8 active\");\n });\n \n test(\"shows CPU warmup state while usage is null\", () => {\ndiff --git a/apps/fabro-web/app/routes/settings-monitoring.tsx b/apps/fabro-web/app/routes/settings-monitoring.tsx\nindex b058e6d27..cde3d733c 100644\n--- a/apps/fabro-web/app/routes/settings-monitoring.tsx\n+++ b/apps/fabro-web/app/routes/settings-monitoring.tsx\n@@ -59,17 +59,14 @@ function RunsPanel() {\n return ;\n }\n \n- const active = info.runs?.active ?? 0;\n+ const slotsUsed = info.runs?.scheduler_slots_used ?? 0;\n const max = settings.server.scheduler.max_concurrent_runs;\n- const percent = max > 0 ? (active / max) * 100 : null;\n+ const percent = max > 0 ? (slotsUsed / max) * 100 : null;\n \n return (\n \n- \n- \n+ \n+ \n \n \n );\ndiff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml\nindex dfa032fe0..c998b3ea2 100644\n--- a/docs/public/api-reference/fabro-api.yaml\n+++ b/docs/public/api-reference/fabro-api.yaml\n@@ -12291,6 +12291,10 @@ components:\n type: integer\n format: int64\n description: Runs currently pending, runnable, or executing.\n+ scheduler_slots_used:\n+ type: integer\n+ format: int64\n+ description: Runs currently occupying scheduler concurrency slots.\n \n SystemResourcesResponse:\n description: Server-visible runtime resource usage for the active Fabro process environment.\ndiff --git a/lib/crates/fabro-server/src/server.rs b/lib/crates/fabro-server/src/server.rs\nindex 0c6c602d0..d224b9991 100644\n--- a/lib/crates/fabro-server/src/server.rs\n+++ b/lib/crates/fabro-server/src/server.rs\n@@ -2467,6 +2467,16 @@ fn compute_queue_positions(runs: &HashMap) -> HashMap bool {\n+ matches!(\n+ status,\n+ RunStatus::Starting\n+ | RunStatus::Running\n+ | RunStatus::Blocked { .. }\n+ | RunStatus::Paused { .. }\n+ )\n+}\n+\n #[allow(\n clippy::result_large_err,\n reason = \"Run ID parsing returns HTTP 400 responses directly.\"\n@@ -4063,15 +4073,7 @@ pub fn spawn_scheduler(state: Arc) {\n let runs = state.runs.lock().expect(\"runs lock poisoned\");\n let active = runs\n .values()\n- .filter(|r| {\n- matches!(\n- r.status,\n- RunStatus::Starting\n- | RunStatus::Running\n- | RunStatus::Blocked { .. }\n- | RunStatus::Paused { .. }\n- )\n- })\n+ .filter(|r| counts_toward_scheduler_capacity(r.status))\n .count();\n let available = state.max_concurrent_runs.saturating_sub(active);\n if available == 0 {\ndiff --git a/lib/crates/fabro-server/src/server/handler/system.rs b/lib/crates/fabro-server/src/server/handler/system.rs\nindex ce836f39b..667d7d81d 100644\n--- a/lib/crates/fabro-server/src/server/handler/system.rs\n+++ b/lib/crates/fabro-server/src/server/handler/system.rs\n@@ -7,9 +7,9 @@ use super::super::{\n BillingByModel, DfParams, FABRO_VERSION, GithubIntegrationStrategy, IntoResponse, Json, Path,\n PruneRunsRequest, PruneRunsResponse, Query, RequiredUser, Response, Router, RunStatus, State,\n StatusCode, SystemInfoResponse, SystemRepairRunIssue, SystemRepairRunsResponse,\n- SystemRunCounts, build_disk_usage_response, build_prune_plan, delete_run_internal, diagnostics,\n- get, post, resolve_interp_string, resource_sampler, spawn_blocking, system_sandbox_provider,\n- to_i64,\n+ SystemRunCounts, build_disk_usage_response, build_prune_plan, counts_toward_scheduler_capacity,\n+ delete_run_internal, diagnostics, get, post, resolve_interp_string, resource_sampler,\n+ spawn_blocking, system_sandbox_provider, to_i64,\n };\n \n pub(super) fn routes() -> Router> {\n@@ -44,7 +44,7 @@ async fn get_server_settings(_auth: RequiredUser, State(state): State>) -> Response {\n let manifest_run_settings = state.manifest_run_settings();\n let server_settings = state.server_settings();\n- let (total_runs, active_runs) = {\n+ let (total_runs, active_runs, scheduler_slots_used) = {\n let runs = state.runs.lock().expect(\"runs lock poisoned\");\n let active = runs\n .values()\n@@ -60,7 +60,11 @@ async fn get_system_info(_auth: RequiredUser, State(state): State>\n )\n })\n .count();\n- (runs.len(), active)\n+ let scheduler_slots_used = runs\n+ .values()\n+ .filter(|run| counts_toward_scheduler_capacity(run.status))\n+ .count();\n+ (runs.len(), active, scheduler_slots_used)\n };\n \n let response = SystemInfoResponse {\n@@ -75,8 +79,9 @@ async fn get_system_info(_auth: RequiredUser, State(state): State>\n storage_dir: Some(state.server_storage_dir().display().to_string()),\n uptime_secs: Some(to_i64(state.started_at.elapsed().as_secs())),\n runs: Some(SystemRunCounts {\n- total: Some(to_i64(total_runs)),\n- active: Some(to_i64(active_runs)),\n+ total: Some(to_i64(total_runs)),\n+ active: Some(to_i64(active_runs)),\n+ scheduler_slots_used: Some(to_i64(scheduler_slots_used)),\n }),\n sandbox_provider: Some(system_sandbox_provider(&manifest_run_settings)),\n };\ndiff --git a/lib/crates/fabro-server/src/server/tests.rs b/lib/crates/fabro-server/src/server/tests.rs\nindex b8ba80cbd..1c2a81375 100644\n--- a/lib/crates/fabro-server/src/server/tests.rs\n+++ b/lib/crates/fabro-server/src/server/tests.rs\n@@ -9513,6 +9513,20 @@ async fn worker_started_child_run_requires_approval_before_becoming_runnable() {\n Some(\"pending\")\n );\n \n+ let response = app\n+ .clone()\n+ .oneshot(bearer_request(\n+ Method::GET,\n+ \"/system/info\",\n+ &user_jwt,\n+ Body::empty(),\n+ ))\n+ .await\n+ .unwrap();\n+ let info_body = response_json!(response, StatusCode::OK).await;\n+ assert_eq!(info_body[\"runs\"][\"active\"], 1);\n+ assert_eq!(info_body[\"runs\"][\"scheduler_slots_used\"], 0);\n+\n {\n let runs = state.runs.lock().expect(\"runs lock poisoned\");\n assert_eq!(\n@@ -13052,6 +13066,31 @@ async fn queue_position_reported_for_runnable_runs() {\n assert_eq!(positions.get(&second_id).copied(), Some(2));\n }\n \n+#[test]\n+fn scheduler_capacity_counts_only_runs_occupying_slots() {\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Submitted));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Pending {\n+ reason: PendingReason::ApprovalRequired,\n+ }));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Runnable));\n+ assert!(counts_toward_scheduler_capacity(RunStatus::Starting));\n+ assert!(counts_toward_scheduler_capacity(RunStatus::Running));\n+ assert!(counts_toward_scheduler_capacity(RunStatus::Blocked {\n+ blocked_reason: BlockedReason::HumanInputRequired,\n+ }));\n+ assert!(counts_toward_scheduler_capacity(RunStatus::Paused {\n+ prior_block: None,\n+ }));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Removing));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Succeeded {\n+ reason: SuccessReason::Completed,\n+ }));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Failed {\n+ reason: FailureReason::WorkflowError,\n+ }));\n+ assert!(!counts_toward_scheduler_capacity(RunStatus::Dead));\n+}\n+\n #[tokio::test(flavor = \"multi_thread\", worker_threads = 2)]\n async fn concurrency_limit_respected() {\n let state = test_app_state_with_options(default_test_server_settings(), RunLayer::default(), 1);\ndiff --git a/lib/crates/fabro-server/tests/it/api/system.rs b/lib/crates/fabro-server/tests/it/api/system.rs\nindex a798d5c26..e7b2abbe0 100644\n--- a/lib/crates/fabro-server/tests/it/api/system.rs\n+++ b/lib/crates/fabro-server/tests/it/api/system.rs\n@@ -266,6 +266,16 @@ async fn test_app_state_with_options_respects_max_concurrent_runs() {\n .is_some_and(std::vec::Vec::is_empty),\n \"second run should still be waiting for scheduler capacity while the first waits at the human gate: {second_questions}\"\n );\n+\n+ let request = Request::builder()\n+ .method(\"GET\")\n+ .uri(api(\"/system/info\"))\n+ .body(Body::empty())\n+ .unwrap();\n+ let response = app.oneshot(request).await.unwrap();\n+ let body = response_json(response, StatusCode::OK, \"GET /api/v1/system/info\").await;\n+ assert_eq!(body[\"runs\"][\"active\"], 2);\n+ assert_eq!(body[\"runs\"][\"scheduler_slots_used\"], 1);\n }\n \n #[tokio::test(flavor = \"multi_thread\", worker_threads = 2)]\ndiff --git a/lib/packages/fabro-api-client/src/models/system-run-counts.ts b/lib/packages/fabro-api-client/src/models/system-run-counts.ts\nindex ad6854320..3fe29ae9b 100644\n--- a/lib/packages/fabro-api-client/src/models/system-run-counts.ts\n+++ b/lib/packages/fabro-api-client/src/models/system-run-counts.ts\n@@ -26,4 +26,8 @@ export interface SystemRunCounts {\n * Runs currently pending, runnable, or executing.\n */\n 'active'?: number;\n+ /**\n+ * Runs currently occupying scheduler concurrency slots.\n+ */\n+ 'scheduler_slots_used'?: number;\n }\n", + "summary": { + "files_changed": 25, + "additions": 1009, + "deletions": 223 + } + } } ], - "conclusion": null, + "conclusion": { + "timestamp": "2026-05-25T23:01:20.471529Z", + "status": "succeeded", + "timing": { + "wall_time_ms": 4598645, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "final_git_commit_sha": "f6d3ed18abc5fc51efb23a3aaefbe97919540988", + "stages": [ + { + "stage_id": "start", + "stage_label": "start", + "timing": { + "wall_time_ms": 0, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "retries": 0 + }, + { + "stage_id": "toolchain", + "stage_label": "toolchain", + "timing": { + "wall_time_ms": 1437, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "retries": 0 + }, + { + "stage_id": "preflight_compile", + "stage_label": "preflight_compile", + "timing": { + "wall_time_ms": 123919, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "retries": 0 + }, + { + "stage_id": "preflight_lint", + "stage_label": "preflight_lint", + "timing": { + "wall_time_ms": 138148, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "retries": 0 + }, + { + "stage_id": "implement", + "stage_label": "implement", + "timing": { + "wall_time_ms": 2153930, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "billing_usd_micros": 17743163, + "retries": 0 + }, + { + "stage_id": "simplify_opus", + "stage_label": "simplify_opus", + "timing": { + "wall_time_ms": 1371402, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "billing_usd_micros": 17181473, + "retries": 0 + }, + { + "stage_id": "simplify_gpt", + "stage_label": "simplify_gpt", + "timing": { + "wall_time_ms": 240238, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "billing_usd_micros": 4885902, + "retries": 0 + }, + { + "stage_id": "verify", + "stage_label": "verify", + "timing": { + "wall_time_ms": 539985, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "retries": 0 + } + ], + "billing": { + "input_tokens": 3842732, + "output_tokens": 62524, + "total_tokens": 26753214, + "reasoning_tokens": 10778, + "cache_read_tokens": 21580482, + "cache_write_tokens": 1256698, + "total_usd_micros": 39810538 + }, + "total_retries": 0, + "diff": {} + }, "sandbox": { "provider": "daytona", "snapshot": "fabro-v12", @@ -2397,6 +2519,40 @@ }, "state": "succeeded" }, + "exit@1": { + "first_event_seq": 1832, + "prompt": null, + "response": null, + "completion": { + "outcome": "succeeded", + "notes": null, + "failure_reason": null, + "timestamp": "2026-05-25T23:01:20.422027Z" + }, + "provider_used": null, + "diff": null, + "script_invocation": null, + "script_timing": null, + "parallel_results": null, + "output": null, + "started_at": "2026-05-25T23:01:20.422004Z", + "handler": "exit", + "timing": { + "wall_time_ms": 0, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, + "usage": { + "input_tokens": 0, + "output_tokens": 0, + "total_tokens": 0, + "reasoning_tokens": 0, + "cache_read_tokens": 0, + "cache_write_tokens": 0 + }, + "state": "succeeded" + }, "toolchain@1": { "first_event_seq": 21, "prompt": null, @@ -2497,7 +2653,12 @@ "first_event_seq": 1822, "prompt": null, "response": null, - "completion": null, + "completion": { + "outcome": "succeeded", + "notes": "Script completed: git fetch origin main 2>&1 && git merge --no-edit --no-stat origin/main 2>&1 && cargo +nightly-2026-04-14 fmt --all 2>&1 && cargo dev docs refresh 2>&1 && cargo +nightly-2026-04-14 fmt --check --all 2>&1 && { command -v rg >/dev/null 2>&1 || { echo 'rg is required for verify'; exit 127; }; } && ! rg -n 'AuthMode::Disabled|RunAuthMethod|RunSubjectProvenance|\\bActorRef\\b|\\bActorKind\\b|AuthenticatedSubject|AuthenticatedService|AuthorizeRunScoped|AuthorizeRunBlob|AuthorizeStageArtifact|AuthorizeCommandLog|auth_method\\s*==\\s*\"disabled\"' lib/crates apps lib/packages docs/public/api-reference/fabro-api.yaml 2>&1 && cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --workspace --status-level slow --profile ci 2>&1 && cargo dev docs check 2>&1 && bun install --frozen-lockfile 2>&1 && (cd apps/fabro-web && bun run typecheck) 2>&1 && (cd apps/fabro-web && bun run test) 2>&1 && (cd lib/packages/fabro-api-client && bun run typecheck) 2>&1 && cargo dev build -- -p fabro-cli --release 2>&1", + "failure_reason": null, + "timestamp": "2026-05-25T23:01:16.016235Z" + }, "provider_used": null, "diff": null, "script_invocation": { @@ -2505,11 +2666,27 @@ "command": "exec 2>&1\ngit fetch origin main 2>&1 && git merge --no-edit --no-stat origin/main 2>&1 && cargo +nightly-2026-04-14 fmt --all 2>&1 && cargo dev docs refresh 2>&1 && cargo +nightly-2026-04-14 fmt --check --all 2>&1 && { command -v rg >/dev/null 2>&1 || { echo 'rg is required for verify'; exit 127; }; } && ! rg -n 'AuthMode::Disabled|RunAuthMethod|RunSubjectProvenance|\\bActorRef\\b|\\bActorKind\\b|AuthenticatedSubject|AuthenticatedService|AuthorizeRunScoped|AuthorizeRunBlob|AuthorizeStageArtifact|AuthorizeCommandLog|auth_method\\s*==\\s*\"disabled\"' lib/crates apps lib/packages docs/public/api-reference/fabro-api.yaml 2>&1 && cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --workspace --status-level slow --profile ci 2>&1 && cargo dev docs check 2>&1 && bun install --frozen-lockfile 2>&1 && (cd apps/fabro-web && bun run typecheck) 2>&1 && (cd apps/fabro-web && bun run test) 2>&1 && (cd lib/packages/fabro-api-client && bun run typecheck) 2>&1 && cargo dev build -- -p fabro-cli --release 2>&1", "language": "shell" }, - "script_timing": null, + "script_timing": { + "output": "blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d", + "exit_code": 0, + "duration_ms": 539971, + "termination": "exited", + "output_bytes": 203867, + "live_streaming": true + }, "parallel_results": null, "output": null, + "output_bytes": 203867, + "live_streaming": true, + "termination": "exited", "started_at": "2026-05-25T22:52:16.026956Z", "handler": "command", + "timing": { + "wall_time_ms": 539985, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, "usage": { "input_tokens": 0, "output_tokens": 0, @@ -2518,7 +2695,7 @@ "cache_read_tokens": 0, "cache_write_tokens": 0 }, - "state": "running" + "state": "succeeded" } } } \ No newline at end of file diff --git a/stages/008-verify@1/diff.patch b/stages/008-verify@1/diff.patch new file mode 100644 index 000000000..156598de8 --- /dev/null +++ b/stages/008-verify@1/diff.patch @@ -0,0 +1,244 @@ +diff --git a/apps/fabro-web/app/routes/settings-monitoring.test.tsx b/apps/fabro-web/app/routes/settings-monitoring.test.tsx +index 10227b98a..070276f8d 100644 +--- a/apps/fabro-web/app/routes/settings-monitoring.test.tsx ++++ b/apps/fabro-web/app/routes/settings-monitoring.test.tsx +@@ -106,7 +106,7 @@ function sampleServerSettings(maxConcurrentRuns = 8): ServerSettings { + describe("SettingsMonitoring route", () => { + beforeEach(() => { + teardownReactTestEnv = setupReactTestEnv(); +- systemInfo = { runs: { active: 3, total: 12 } }; ++ systemInfo = { runs: { active: 3, scheduler_slots_used: 1, total: 12 } }; + serverSettings = sampleServerSettings(); + }); + +@@ -133,7 +133,8 @@ describe("SettingsMonitoring route", () => { + expect(text).toContain("5s"); + expect(text).toContain("3 GiB"); + expect(text).toContain("8 GiB"); +- expect(text).toContain("3 / 8 active"); ++ expect(text).toContain("1 / 8 slots used"); ++ expect(text).not.toContain("3 / 8 active"); + }); + + test("shows CPU warmup state while usage is null", () => { +diff --git a/apps/fabro-web/app/routes/settings-monitoring.tsx b/apps/fabro-web/app/routes/settings-monitoring.tsx +index b058e6d27..cde3d733c 100644 +--- a/apps/fabro-web/app/routes/settings-monitoring.tsx ++++ b/apps/fabro-web/app/routes/settings-monitoring.tsx +@@ -59,17 +59,14 @@ function RunsPanel() { + return ; + } + +- const active = info.runs?.active ?? 0; ++ const slotsUsed = info.runs?.scheduler_slots_used ?? 0; + const max = settings.server.scheduler.max_concurrent_runs; +- const percent = max > 0 ? (active / max) * 100 : null; ++ const percent = max > 0 ? (slotsUsed / max) * 100 : null; + + return ( + +- +- ++ ++ + + + ); +diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml +index dfa032fe0..c998b3ea2 100644 +--- a/docs/public/api-reference/fabro-api.yaml ++++ b/docs/public/api-reference/fabro-api.yaml +@@ -12291,6 +12291,10 @@ components: + type: integer + format: int64 + description: Runs currently pending, runnable, or executing. ++ scheduler_slots_used: ++ type: integer ++ format: int64 ++ description: Runs currently occupying scheduler concurrency slots. + + SystemResourcesResponse: + description: Server-visible runtime resource usage for the active Fabro process environment. +diff --git a/lib/crates/fabro-server/src/server.rs b/lib/crates/fabro-server/src/server.rs +index 0c6c602d0..d224b9991 100644 +--- a/lib/crates/fabro-server/src/server.rs ++++ b/lib/crates/fabro-server/src/server.rs +@@ -2467,6 +2467,16 @@ fn compute_queue_positions(runs: &HashMap) -> HashMap bool { ++ matches!( ++ status, ++ RunStatus::Starting ++ | RunStatus::Running ++ | RunStatus::Blocked { .. } ++ | RunStatus::Paused { .. } ++ ) ++} ++ + #[allow( + clippy::result_large_err, + reason = "Run ID parsing returns HTTP 400 responses directly." +@@ -4063,15 +4073,7 @@ pub fn spawn_scheduler(state: Arc) { + let runs = state.runs.lock().expect("runs lock poisoned"); + let active = runs + .values() +- .filter(|r| { +- matches!( +- r.status, +- RunStatus::Starting +- | RunStatus::Running +- | RunStatus::Blocked { .. } +- | RunStatus::Paused { .. } +- ) +- }) ++ .filter(|r| counts_toward_scheduler_capacity(r.status)) + .count(); + let available = state.max_concurrent_runs.saturating_sub(active); + if available == 0 { +diff --git a/lib/crates/fabro-server/src/server/handler/system.rs b/lib/crates/fabro-server/src/server/handler/system.rs +index ce836f39b..667d7d81d 100644 +--- a/lib/crates/fabro-server/src/server/handler/system.rs ++++ b/lib/crates/fabro-server/src/server/handler/system.rs +@@ -7,9 +7,9 @@ use super::super::{ + BillingByModel, DfParams, FABRO_VERSION, GithubIntegrationStrategy, IntoResponse, Json, Path, + PruneRunsRequest, PruneRunsResponse, Query, RequiredUser, Response, Router, RunStatus, State, + StatusCode, SystemInfoResponse, SystemRepairRunIssue, SystemRepairRunsResponse, +- SystemRunCounts, build_disk_usage_response, build_prune_plan, delete_run_internal, diagnostics, +- get, post, resolve_interp_string, resource_sampler, spawn_blocking, system_sandbox_provider, +- to_i64, ++ SystemRunCounts, build_disk_usage_response, build_prune_plan, counts_toward_scheduler_capacity, ++ delete_run_internal, diagnostics, get, post, resolve_interp_string, resource_sampler, ++ spawn_blocking, system_sandbox_provider, to_i64, + }; + + pub(super) fn routes() -> Router> { +@@ -44,7 +44,7 @@ async fn get_server_settings(_auth: RequiredUser, State(state): State>) -> Response { + let manifest_run_settings = state.manifest_run_settings(); + let server_settings = state.server_settings(); +- let (total_runs, active_runs) = { ++ let (total_runs, active_runs, scheduler_slots_used) = { + let runs = state.runs.lock().expect("runs lock poisoned"); + let active = runs + .values() +@@ -60,7 +60,11 @@ async fn get_system_info(_auth: RequiredUser, State(state): State> + ) + }) + .count(); +- (runs.len(), active) ++ let scheduler_slots_used = runs ++ .values() ++ .filter(|run| counts_toward_scheduler_capacity(run.status)) ++ .count(); ++ (runs.len(), active, scheduler_slots_used) + }; + + let response = SystemInfoResponse { +@@ -75,8 +79,9 @@ async fn get_system_info(_auth: RequiredUser, State(state): State> + storage_dir: Some(state.server_storage_dir().display().to_string()), + uptime_secs: Some(to_i64(state.started_at.elapsed().as_secs())), + runs: Some(SystemRunCounts { +- total: Some(to_i64(total_runs)), +- active: Some(to_i64(active_runs)), ++ total: Some(to_i64(total_runs)), ++ active: Some(to_i64(active_runs)), ++ scheduler_slots_used: Some(to_i64(scheduler_slots_used)), + }), + sandbox_provider: Some(system_sandbox_provider(&manifest_run_settings)), + }; +diff --git a/lib/crates/fabro-server/src/server/tests.rs b/lib/crates/fabro-server/src/server/tests.rs +index b8ba80cbd..1c2a81375 100644 +--- a/lib/crates/fabro-server/src/server/tests.rs ++++ b/lib/crates/fabro-server/src/server/tests.rs +@@ -9513,6 +9513,20 @@ async fn worker_started_child_run_requires_approval_before_becoming_runnable() { + Some("pending") + ); + ++ let response = app ++ .clone() ++ .oneshot(bearer_request( ++ Method::GET, ++ "/system/info", ++ &user_jwt, ++ Body::empty(), ++ )) ++ .await ++ .unwrap(); ++ let info_body = response_json!(response, StatusCode::OK).await; ++ assert_eq!(info_body["runs"]["active"], 1); ++ assert_eq!(info_body["runs"]["scheduler_slots_used"], 0); ++ + { + let runs = state.runs.lock().expect("runs lock poisoned"); + assert_eq!( +@@ -13052,6 +13066,31 @@ async fn queue_position_reported_for_runnable_runs() { + assert_eq!(positions.get(&second_id).copied(), Some(2)); + } + ++#[test] ++fn scheduler_capacity_counts_only_runs_occupying_slots() { ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Submitted)); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Pending { ++ reason: PendingReason::ApprovalRequired, ++ })); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Runnable)); ++ assert!(counts_toward_scheduler_capacity(RunStatus::Starting)); ++ assert!(counts_toward_scheduler_capacity(RunStatus::Running)); ++ assert!(counts_toward_scheduler_capacity(RunStatus::Blocked { ++ blocked_reason: BlockedReason::HumanInputRequired, ++ })); ++ assert!(counts_toward_scheduler_capacity(RunStatus::Paused { ++ prior_block: None, ++ })); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Removing)); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Succeeded { ++ reason: SuccessReason::Completed, ++ })); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Failed { ++ reason: FailureReason::WorkflowError, ++ })); ++ assert!(!counts_toward_scheduler_capacity(RunStatus::Dead)); ++} ++ + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn concurrency_limit_respected() { + let state = test_app_state_with_options(default_test_server_settings(), RunLayer::default(), 1); +diff --git a/lib/crates/fabro-server/tests/it/api/system.rs b/lib/crates/fabro-server/tests/it/api/system.rs +index a798d5c26..e7b2abbe0 100644 +--- a/lib/crates/fabro-server/tests/it/api/system.rs ++++ b/lib/crates/fabro-server/tests/it/api/system.rs +@@ -266,6 +266,16 @@ async fn test_app_state_with_options_respects_max_concurrent_runs() { + .is_some_and(std::vec::Vec::is_empty), + "second run should still be waiting for scheduler capacity while the first waits at the human gate: {second_questions}" + ); ++ ++ let request = Request::builder() ++ .method("GET") ++ .uri(api("/system/info")) ++ .body(Body::empty()) ++ .unwrap(); ++ let response = app.oneshot(request).await.unwrap(); ++ let body = response_json(response, StatusCode::OK, "GET /api/v1/system/info").await; ++ assert_eq!(body["runs"]["active"], 2); ++ assert_eq!(body["runs"]["scheduler_slots_used"], 1); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] +diff --git a/lib/packages/fabro-api-client/src/models/system-run-counts.ts b/lib/packages/fabro-api-client/src/models/system-run-counts.ts +index ad6854320..3fe29ae9b 100644 +--- a/lib/packages/fabro-api-client/src/models/system-run-counts.ts ++++ b/lib/packages/fabro-api-client/src/models/system-run-counts.ts +@@ -26,4 +26,8 @@ export interface SystemRunCounts { + * Runs currently pending, runnable, or executing. + */ + 'active'?: number; ++ /** ++ * Runs currently occupying scheduler concurrency slots. ++ */ ++ 'scheduler_slots_used'?: number; + } diff --git a/stages/008-verify@1/output.log b/stages/008-verify@1/output.log new file mode 100644 index 000000000..f49325f8f --- /dev/null +++ b/stages/008-verify@1/output.log @@ -0,0 +1 @@ +blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d \ No newline at end of file diff --git a/stages/008-verify@1/script_timing.json b/stages/008-verify@1/script_timing.json new file mode 100644 index 000000000..82ebfda3b --- /dev/null +++ b/stages/008-verify@1/script_timing.json @@ -0,0 +1,8 @@ +{ + "output": "blob://sha256/dadf10e5135a6e1f3f0d7a88bd81fbc3cb82b8de6434e903365683528903553d", + "exit_code": 0, + "duration_ms": 539971, + "termination": "exited", + "output_bytes": 203867, + "live_streaming": true +} \ No newline at end of file diff --git a/stages/008-verify@1/status.json b/stages/008-verify@1/status.json new file mode 100644 index 000000000..ae9a655de --- /dev/null +++ b/stages/008-verify@1/status.json @@ -0,0 +1,6 @@ +{ + "outcome": "succeeded", + "notes": "Script completed: git fetch origin main 2>&1 && git merge --no-edit --no-stat origin/main 2>&1 && cargo +nightly-2026-04-14 fmt --all 2>&1 && cargo dev docs refresh 2>&1 && cargo +nightly-2026-04-14 fmt --check --all 2>&1 && { command -v rg >/dev/null 2>&1 || { echo 'rg is required for verify'; exit 127; }; } && ! rg -n 'AuthMode::Disabled|RunAuthMethod|RunSubjectProvenance|\\bActorRef\\b|\\bActorKind\\b|AuthenticatedSubject|AuthenticatedService|AuthorizeRunScoped|AuthorizeRunBlob|AuthorizeStageArtifact|AuthorizeCommandLog|auth_method\\s*==\\s*\"disabled\"' lib/crates apps lib/packages docs/public/api-reference/fabro-api.yaml 2>&1 && cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --workspace --status-level slow --profile ci 2>&1 && cargo dev docs check 2>&1 && bun install --frozen-lockfile 2>&1 && (cd apps/fabro-web && bun run typecheck) 2>&1 && (cd apps/fabro-web && bun run test) 2>&1 && (cd lib/packages/fabro-api-client && bun run typecheck) 2>&1 && cargo dev build -- -p fabro-cli --release 2>&1", + "failure_reason": null, + "timestamp": "2026-05-25T23:01:16.016235Z" +} \ No newline at end of file diff --git a/stages/009-exit@1/status.json b/stages/009-exit@1/status.json new file mode 100644 index 000000000..b271d98a3 --- /dev/null +++ b/stages/009-exit@1/status.json @@ -0,0 +1,6 @@ +{ + "outcome": "succeeded", + "notes": null, + "failure_reason": null, + "timestamp": "2026-05-25T23:01:20.422027Z" +} \ No newline at end of file