- {formatTokens(row.value)}
+ {buckets.map((bucket) => (
+
+ {bucket.label}
+
+ {formatTokens(bucket.value)}
))}
@@ -128,42 +105,16 @@ function TokenBreakdown({
* Renders an `input / output` token count. When the row has model usage,
* hovering the count reveals the cache breakdown.
*/
-function TokensCell({
- inputTokens,
- outputTokens,
- cacheReadTokens,
- cacheWriteTokens,
-}: {
- inputTokens: number | null;
- outputTokens: number | null;
- cacheReadTokens: number | null;
- cacheWriteTokens: number | null;
-}) {
+function TokensCell({ billing }: { billing: BilledTokenCounts | null }) {
const display = (
<>
- {formatTokens(inputTokens)} / {" "}
- {formatTokens(outputTokens)}
+ {formatTokens(billing?.input_tokens)} / {" "}
+ {formatTokens(billing ? billableOutputTokens(billing) : null)}
>
);
- if (
- inputTokens == null ||
- outputTokens == null ||
- cacheReadTokens == null ||
- cacheWriteTokens == null
- ) {
- return display;
- }
+ if (!billing) return display;
return (
-
- }
- >
+ }>
{display}
);
@@ -189,15 +140,14 @@ export default function RunBilling({ params }: { params: { id: string } }) {
if (!billing) return [];
return billing.by_model
.map((entry) => ({
- model: formatModelRef(entry.model) ?? EMPTY_VALUE,
- stages: entry.stages,
- inputTokens: entry.billing.input_tokens,
- outputTokens: entry.billing.output_tokens + entry.billing.reasoning_tokens,
- cacheReadTokens: entry.billing.cache_read_tokens,
- cacheWriteTokens: entry.billing.cache_write_tokens,
- totalUsdMicros: entry.billing.total_usd_micros,
+ model: formatModelRef(entry.model) ?? EMPTY_VALUE,
+ stages: entry.stages,
+ billing: entry.billing,
}))
- .sort((a, b) => (b.totalUsdMicros ?? -1) - (a.totalUsdMicros ?? -1));
+ .sort(
+ (a, b) =>
+ (b.billing.total_usd_micros ?? -1) - (a.billing.total_usd_micros ?? -1),
+ );
}, [billing]);
// Re-derive only the in-flight rows on each tick; everything else stays put.
@@ -218,16 +168,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
: (billing?.totals.timing.wall_time_ms ?? 0);
const hasLlmStages = (billing?.by_model.length ?? 0) > 0;
- const totalInput = hasLlmStages ? (billing?.totals.input_tokens ?? null) : null;
- const totalOutput = hasLlmStages && billing
- ? billing.totals.output_tokens + billing.totals.reasoning_tokens
- : null;
- const totalCacheRead = hasLlmStages
- ? (billing?.totals.cache_read_tokens ?? null)
- : null;
- const totalCacheWrite = hasLlmStages
- ? (billing?.totals.cache_write_tokens ?? null)
- : null;
+ const totalBilling = hasLlmStages && billing ? billing.totals : null;
const totalUsdMicros = billing?.totals.total_usd_micros;
const modelStageCount = modelBreakdown.reduce((sum, row) => sum + row.stages, 0);
const visibleRows = rows.filter(isVisibleRow);
@@ -268,18 +209,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.model ?? EMPTY_VALUE}
-
+
{formatDurationMs(row.wallTimeMs)}
- {formatUsdMicrosOrDash(row.totalUsdMicros)}
+ {formatUsdMicrosOrDash(row.billing?.total_usd_micros)}
))}
@@ -289,12 +225,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
Total
All models
-
+
{formatDurationMs(totalWallTimeMs)}
@@ -328,15 +259,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.stages}
-
+
- {formatUsdMicrosOrDash(row.totalUsdMicros)}
+ {formatUsdMicrosOrDash(row.billing.total_usd_micros)}
))}
@@ -348,12 +274,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{modelStageCount}
-
+
{formatUsdMicrosOrDash(totalUsdMicros)}
diff --git a/apps/fabro-web/app/routes/run-detail.tsx b/apps/fabro-web/app/routes/run-detail.tsx
index 2e8f1a44e..b960fe95e 100644
--- a/apps/fabro-web/app/routes/run-detail.tsx
+++ b/apps/fabro-web/app/routes/run-detail.tsx
@@ -19,6 +19,7 @@ import {
} from "../components/ui";
import { mutateRunListCaches } from "../lib/board-cache";
import { useDemoMode } from "../lib/demo-mode";
+import { useTickingNow } from "../lib/time";
import { useSWRConfig } from "swr";
import {
useArchiveRun,
@@ -56,10 +57,7 @@ import {
lifecycleActionVisibility,
updateLifecycleToastState,
} from "./run-detail/lifecycle-toasts";
-import {
- buildRunDetailRun,
- useTickingNow,
-} from "./run-detail/model";
+import { buildRunDetailRun } from "./run-detail/model";
import {
buildRunDetailTabs,
childRouteLayoutFlags,
@@ -69,6 +67,8 @@ import {
export const handle = { hideHeader: true };
+const RUN_TIMING_REFRESH_INTERVAL_MS = 30_000;
+
type LifecycleTrigger = () => Promise;
export interface DockMeasurement {
@@ -95,7 +95,7 @@ export function meta({ data }: any) {
export default function RunDetail({ params }: { params: { id: string } }) {
const demoMode = useDemoMode();
- const runQuery = useRun(params.id);
+ const runQuery = useRun(params.id, RUN_TIMING_REFRESH_INTERVAL_MS);
const runStateQuery = useRunState(params.id);
const summary = runQuery.data;
const run = summary ? buildRunDetailRun(summary) : null;
@@ -135,7 +135,10 @@ export default function RunDetail({ params }: { params: { id: string } }) {
childrenCount,
});
const steerBarRef = useRef(null);
- const now = useTickingNow(30_000);
+ const now = useTickingNow(
+ summary != null && summary.timestamps.completed_at == null,
+ RUN_TIMING_REFRESH_INTERVAL_MS,
+ );
const { fullHeight, hideSteerBar } = childRouteLayoutFlags(matches);
useRunEvents(params.id);
diff --git a/apps/fabro-web/app/routes/run-detail/header.tsx b/apps/fabro-web/app/routes/run-detail/header.tsx
index 70b323156..28135ec68 100644
--- a/apps/fabro-web/app/routes/run-detail/header.tsx
+++ b/apps/fabro-web/app/routes/run-detail/header.tsx
@@ -327,6 +327,7 @@ function DurationPopover({
}) {
const endMs = completedAt != null ? Date.parse(completedAt) : now;
const sinceCreatedMs = Math.max(0, endMs - Date.parse(createdAt));
+ const isRunning = completedAt == null;
return (
<>
Duration
@@ -336,8 +337,14 @@ function DurationPopover({
{formatDurationMs(sinceCreatedMs)}
-
Active (inference + tools)
+
+ Active (inference + tools){isRunning ? " — estimated" : ""}
+
{formatDurationMs(timing.active_time_ms)}
+
+ {formatDurationMs(timing.inference_time_ms)} inference ·{" "}
+ {formatDurationMs(timing.tool_time_ms)} tools
+
>
diff --git a/apps/fabro-web/app/routes/run-detail/model.ts b/apps/fabro-web/app/routes/run-detail/model.ts
index e0b853ce9..4428810e0 100644
--- a/apps/fabro-web/app/routes/run-detail/model.ts
+++ b/apps/fabro-web/app/routes/run-detail/model.ts
@@ -1,6 +1,3 @@
-import { useState } from "react";
-
-import { useInterval } from "../../hooks/effects";
import {
isRunStatus,
mapRunToRunItem,
@@ -8,12 +5,6 @@ import {
type Run,
} from "../../data/runs";
-export function useTickingNow(intervalMs: number): number {
- const [now, setNow] = useState(() => Date.now());
- useInterval(() => setNow(Date.now()), intervalMs);
- return now;
-}
-
export type RunDetailRun = ReturnType & {
statusLabel: string;
statusDot: string;
diff --git a/apps/fabro-web/app/routes/run-stages-chat.test.tsx b/apps/fabro-web/app/routes/run-stages-chat.test.tsx
index c4307ebbc..1142ae6b8 100644
--- a/apps/fabro-web/app/routes/run-stages-chat.test.tsx
+++ b/apps/fabro-web/app/routes/run-stages-chat.test.tsx
@@ -3,6 +3,7 @@ import { renderToStaticMarkup } from "react-dom/server";
import { StageState } from "@qltysh/fabro-api-client";
import type { Stage } from "../lib/stage-sidebar";
+import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { StageChatView } from "./run-stages";
function stage(overrides: Partial = {}): Stage {
@@ -18,6 +19,7 @@ function stage(overrides: Partial = {}): Stage {
resumedFromStageId: null,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
+ billing: makeBilledTokenCounts(),
...overrides,
};
}
diff --git a/apps/fabro-web/app/routes/run-stages-details.test.tsx b/apps/fabro-web/app/routes/run-stages-details.test.tsx
index da18f7e67..4483ec4e9 100644
--- a/apps/fabro-web/app/routes/run-stages-details.test.tsx
+++ b/apps/fabro-web/app/routes/run-stages-details.test.tsx
@@ -1,9 +1,14 @@
import { describe, expect, test } from "bun:test";
import { renderToStaticMarkup } from "react-dom/server";
-import type { ReasoningOutput } from "@qltysh/fabro-api-client";
+import type {
+ BilledTokenCounts,
+ ReasoningOutput,
+ StageModelUsage,
+} from "@qltysh/fabro-api-client";
-import { EventDetails } from "./run-stages";
+import { makeBilledTokenCounts } from "../lib/test-fixtures";
+import { EventDetails, ModelUsagePopover } from "./run-stages";
const RUN_START = "2026-04-09T12:00:00Z";
@@ -72,3 +77,77 @@ describe("EventDetails reasoning", () => {
expect(html).toContain(`${"x".repeat(280)}…`);
});
});
+
+const PROVIDER_USED: StageModelUsage = {
+ mode: "agent",
+ provider: "moonshot",
+ model: "kimi-k3",
+ reasoning_effort: "max",
+};
+
+function popoverMarkup(counts: BilledTokenCounts): string {
+ return renderToStaticMarkup(
+ ,
+ );
+}
+
+describe("ModelUsagePopover billing", () => {
+ test("shows the visit's token buckets and cost next to the model", () => {
+ const html = popoverMarkup(
+ makeBilledTokenCounts({
+ input_tokens: 28_640,
+ output_tokens: 7_550,
+ reasoning_tokens: 1_200,
+ cache_read_tokens: 4_800,
+ cache_write_tokens: 1_500,
+ total_tokens: 43_690,
+ total_usd_micros: 720_000,
+ }),
+ );
+
+ expect(html).toContain("kimi-k3");
+ expect(html).toContain("Cache read");
+ expect(html).toContain("4.8k");
+ expect(html).toContain("Cache creation");
+ expect(html).toContain("1.5k");
+ expect(html).toContain("Uncached");
+ expect(html).toContain("28.6k");
+ // Output folds in reasoning tokens, matching the Billing tab.
+ expect(html).toContain("Output");
+ expect(html).toContain("8.8k");
+ expect(html).toContain("Cost");
+ expect(html).toContain("$0.72");
+ });
+
+ test("omits the token section for a stage that called no model", () => {
+ const html = popoverMarkup(makeBilledTokenCounts());
+
+ expect(html).toContain("kimi-k3");
+ expect(html).not.toContain("Tokens");
+ expect(html).not.toContain("Cost");
+ });
+
+ test("still shows tokens when nothing priced the stage", () => {
+ const html = popoverMarkup(
+ makeBilledTokenCounts({
+ input_tokens: 1_000,
+ output_tokens: 500,
+ total_tokens: 1_500,
+ }),
+ );
+
+ expect(html).toContain("Uncached");
+ expect(html).toContain("1.0k");
+ expect(html).not.toContain("Cost");
+ });
+
+ test("shows a provider-reported cost when token counts are unavailable", () => {
+ const html = popoverMarkup(
+ makeBilledTokenCounts({ total_usd_micros: 720_000 }),
+ );
+
+ expect(html).toContain("kimi-k3");
+ expect(html).toContain("Cost");
+ expect(html).toContain("$0.72");
+ });
+});
diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx
index 43f4410f1..0f03c6a98 100644
--- a/apps/fabro-web/app/routes/run-stages.tsx
+++ b/apps/fabro-web/app/routes/run-stages.tsx
@@ -67,7 +67,9 @@ import {
formatBytes,
formatDurationMs,
formatTokenCount,
+ formatUsdMicros,
} from "../lib/format";
+import { billingTokenBuckets, hasBillingUsage } from "../lib/billing";
import { plural } from "../lib/plural";
import {
useRun,
@@ -93,6 +95,7 @@ import {
type UnknownRecord,
} from "../lib/unknown";
import type {
+ BilledTokenCounts,
EventEnvelope,
ReasoningOutput,
StageHandler,
@@ -866,10 +869,42 @@ export function formatStageModelUsageLabel(
return effort ? `${model}[${effort}]` : model;
}
-function ModelUsagePopover({
+const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
+
+/** Tokens and cost for this stage visit alone. */
+function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
+ if (!hasBillingUsage(billing)) return null;
+ const buckets = billingTokenBuckets(billing);
+ const cost = formatUsdMicros(billing.total_usd_micros);
+ return (
+
+
Tokens
+
+ {buckets.map((bucket) => (
+
+
+ {bucket.value === 0
+ ? "0"
+ : formatTokenCount(bucket.value, { compactDecimal: true })}
+
+
+ ))}
+ {cost && (
+
+ {cost}
+
+ )}
+
+
+ );
+}
+
+export function ModelUsagePopover({
providerUsed,
+ billing,
}: {
providerUsed: StageModelUsage;
+ billing: BilledTokenCounts;
}) {
return (
<>
@@ -892,6 +927,7 @@ function ModelUsagePopover({
{providerUsed.speed}
)}
+
>
);
}
@@ -1905,6 +1941,7 @@ function EventsToolbar({
filteredCount,
totalCount,
providerUsed,
+ billing,
events,
runId,
stageId,
@@ -1924,6 +1961,7 @@ function EventsToolbar({
filteredCount: number;
totalCount: number;
providerUsed: StageModelUsage | null;
+ billing: BilledTokenCounts;
events: EventEnvelope[];
runId: string;
stageId: string;
@@ -2004,7 +2042,9 @@ function EventsToolbar({
className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${
showFilters ? "" : "ml-auto"
}`}
- content={ }
+ content={
+
+ }
>
{modelUsageLabel}
@@ -2359,6 +2399,7 @@ function RunStageActivityStage({
effectiveTab === "primary" ? turns.length : debugEvents.length
}
providerUsed={selectedStage.providerUsed}
+ billing={selectedStage.billing}
events={stageEventsQuery.data ?? []}
runId={runId}
stageId={selectedStageId}
diff --git a/apps/fabro-web/app/routes/runs.tsx b/apps/fabro-web/app/routes/runs.tsx
index 7da0719ea..b1e09bf67 100644
--- a/apps/fabro-web/app/routes/runs.tsx
+++ b/apps/fabro-web/app/routes/runs.tsx
@@ -26,6 +26,7 @@ import { ciConfig, columnForRun, columnStatusDisplay, columnStatuses, deriveCiSt
import type { CiStatus, CheckRun, CheckStatus, RunItem } from "../data/runs";
import { EmptyState } from "../components/state";
import { PullRequestChip } from "../components/pull-request-chip";
+import { SizeChip } from "../components/size-chip";
import {
summarizeBatchLifecycleAction,
} from "../components/runs-list/batch-lifecycle";
@@ -345,7 +346,7 @@ function PrCard({
// All inline footer metadata on PrCard belongs in this one row. Adding a new
// piece as a sibling `` below the card body recreates a recurring bug
-// where stats stack onto separate lines instead of sitting next to elapsed/actions.
+// where stats stack onto separate lines instead of sitting next to size/actions.
function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
const hasActions = actions != null && actions.length > 0;
const hasStats =
@@ -354,7 +355,7 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
(pr.additions != null && pr.additions !== 0) ||
(pr.deletions != null && pr.deletions !== 0);
- if (!hasStats && !hasActions && pr.elapsed == null) return null;
+ if (!hasStats && !hasActions && pr.size == null) return null;
return (
@@ -416,9 +417,9 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
))}
)}
- {pr.elapsed != null && (
-
- {pr.elapsed}
+ {pr.size != null && (
+
+
)}
diff --git a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md
index 90438d3b5..c0980ed01 100644
--- a/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md
+++ b/docs/plans/2026-05-21-wall-and-active-time-metrics-plan.md
@@ -155,9 +155,16 @@ git diff --check
- Inference time is Fabro-observed LLM request/stream elapsed time, not
provider-reported model-only compute time.
-- LLM retry backoff, queueing outside a request/stream, human waits, steering
- waits, and scheduler gaps are wall time but not active time.
-- Active timing is finalized-event based in v1; live active-time ticking can be
- added later if it becomes necessary.
+- Queueing outside a request/stream, human waits, steering waits, and scheduler
+ gaps are wall time but not active time. Retry delay inside an open LLM request
+ bracket follows the executor stopwatch and counts as inference time.
+- ~~Active timing is finalized-event based in v1; live active-time ticking can
+ be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became
+ necessary: a run parked in one long agent stage reported ~12% of its wall time
+ as active, because in-flight stages contributed nothing. Stage projections now
+ accumulate inference and tool brackets from the event log and expose
+ `StageProjection::live_timing(now)`, the active-time twin of
+ `live_wall_time_ms`. Finalized values remain authoritative and still replace
+ the live estimate at terminal events. Implemented in PR #647.
- No compatibility layer is required for existing API clients or stored run
event data.
diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml
index b028d89c0..30c9aa83c 100644
--- a/docs/public/api-reference/fabro-api.yaml
+++ b/docs/public/api-reference/fabro-api.yaml
@@ -10673,7 +10673,38 @@ components:
- type: "null"
description: |
Per-attempt timing breakdown for the latest terminal attempt:
- wall time plus the active inference/tool breakdown.
+ wall time plus the active inference/tool breakdown. Null while the
+ stage is still in flight; the live estimate is derived from
+ `live_inference_ms`, `live_tool_ms`, and any open bracket.
+ live_inference_ms:
+ type: integer
+ format: uint64
+ minimum: 0
+ default: 0
+ description: |
+ Inference time accumulated from closed brackets during the current
+ attempt. Live estimate only — the authoritative value arrives with
+ the terminal event and lands in `timing`. Excludes the currently
+ open bracket, whose span is measured from `inference.started_at`.
+ example: 78230
+ live_tool_ms:
+ type: integer
+ format: uint64
+ minimum: 0
+ default: 0
+ description: |
+ Tool time accumulated from closed tool batches during the current
+ attempt. A batch spans the first dispatched call through the
+ completion that drains the last outstanding one, so tools running
+ concurrently within a turn are counted once.
+ example: 7588
+ tool_batch:
+ oneOf:
+ - $ref: "#/components/schemas/StageToolBatchProjection"
+ - type: "null"
+ description: |
+ Open tool batch: when the batch started and which calls have not
+ yet reported completion.
usage:
$ref: "#/components/schemas/BilledTokenCounts"
model:
@@ -10727,6 +10758,13 @@ components:
Open inference bracket, if the event log contains one. Present means
a model request was dispatched and no closing event has been seen —
not that the model is computing right now.
+ acp_started_at:
+ type: ["string", "null"]
+ format: date-time
+ description: >
+ Start of an external ACP agent process, if one is running. ACP
+ agents do not expose Fabro's internal LLM brackets, so the process
+ lifetime supplies their live inference estimate.
agent_control:
$ref: "#/components/schemas/AgentControlState"
description: Whether the agent is executing normally or waiting for steering after an interrupt.
@@ -10734,6 +10772,37 @@ components:
$ref: "#/components/schemas/StageState"
description: Lifecycle state of the stage projection.
+ StageToolBatchProjection:
+ description: >
+ One open tool batch: tool calls dispatched together that have not all
+ reported completion. `open_call_ids` is a set rather than a count so a
+ duplicated completion in a replayed log cannot drain the batch early.
+ type: object
+ required:
+ - session_id
+ - started_at
+ - open_call_ids
+ properties:
+ session_id:
+ type: string
+ description: >
+ Root agent session that dispatched the batch. Transitions are
+ gated on it so delayed events from a replaced session cannot
+ mutate the current batch.
+ started_at:
+ type: string
+ format: date-time
+ description: >
+ When the batch opened — the first dispatched call observed while no
+ other calls were outstanding.
+ open_call_ids:
+ type: array
+ minItems: 1
+ uniqueItems: true
+ items:
+ type: string
+ description: Calls dispatched but not yet completed, by tool call id.
+
StageInferenceProjection:
description: >
One open inference bracket: a dispatched LLM request that has not yet
@@ -12244,9 +12313,17 @@ components:
observed LLM request/stream elapsed time; `tool_time_ms` is tool or
command execution elapsed time; `active_time_ms` equals
`inference_time_ms + tool_time_ms`.
+
+ For a terminal stage these come from the worker's own stopwatch and are
+ authoritative. For a stage still in flight they are a live estimate
+ reconstructed from the event log, and `active_time_ms` is clamped to
+ `wall_time_ms`. The estimate is replaced by the authoritative
+ breakdown when the stage reaches a terminal event.
type: object
required:
- wall_time_ms
+ - inference_time_ms
+ - tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@@ -12278,9 +12355,16 @@ components:
Timing rollup for an entire run. Active fields sum work across stage
visits, so `active_time_ms` can exceed `wall_time_ms` when parallel
branches run concurrently.
+
+ For a running run, stages still in flight contribute a live estimate
+ rather than nothing, so wall and active both advance continuously.
+ Unlike `StageTiming`, active is not clamped to wall here — concurrent
+ branches can legitimately sum past run wall time.
type: object
required:
- wall_time_ms
+ - inference_time_ms
+ - tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@@ -12609,6 +12693,7 @@ components:
- status
- node_id
- visit
+ - billing
properties:
id:
$ref: "#/components/schemas/StageId"
@@ -12668,6 +12753,15 @@ components:
format: date-time
description: Wall-clock time the latest attempt of this stage started, if known.
example: "2026-04-29T12:34:56Z"
+ billing:
+ $ref: "#/components/schemas/BilledTokenCounts"
+ description: >-
+ Token counts for this stage execution alone. `total_usd_micros` is
+ the provider-reported cost when there is one, otherwise the server
+ catalog's price for these tokens — the same pricing the
+ `/runs/{id}/billing` rows use. All-zero counts mean the stage made
+ no model calls. Unlike the billing rows, which sum every visit of a
+ node, this covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────
diff --git a/docs/public/reference/dot-language.mdx b/docs/public/reference/dot-language.mdx
index 8c350d1ba..ee44c3668 100644
--- a/docs/public/reference/dot-language.mdx
+++ b/docs/public/reference/dot-language.mdx
@@ -113,7 +113,7 @@ plan [label="Plan", prompt="Create an implementation plan."]
**Node identifiers** must start with a letter or underscore, followed by letters, digits, or underscores (e.g. `run_tests`, `gate_1`, `_private`).
-Nodes referenced in edges are auto-created if not explicitly declared.
+Every node used by an edge needs its own declaration. Validation fails when an edge names a node the workflow never declares, because that is nearly always a typo or a rename that missed an edge. The declaration can come before or after the edges that use it, and it can live in a subgraph.
### Edge declarations
diff --git a/lib/apps/fabro-cli/src/commands/run/create.rs b/lib/apps/fabro-cli/src/commands/run/create.rs
index ad4360384..159160d7e 100644
--- a/lib/apps/fabro-cli/src/commands/run/create.rs
+++ b/lib/apps/fabro-cli/src/commands/run/create.rs
@@ -61,11 +61,8 @@ pub(crate) async fn create_run(
None
};
- let mut validation = manifest_validation::validate_manifest(
- &RunLayer::default(),
- &built.manifest,
- ctx.catalog()?,
- )?;
+ let mut validation =
+ manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
let diagnostics = api_diagnostics_to_local(&validation.workflow.diagnostics);
if !quiet {
diff --git a/lib/apps/fabro-cli/src/commands/run/runner.rs b/lib/apps/fabro-cli/src/commands/run/runner.rs
index 4e6b72b8b..13f0fc9a6 100644
--- a/lib/apps/fabro-cli/src/commands/run/runner.rs
+++ b/lib/apps/fabro-cli/src/commands/run/runner.rs
@@ -110,7 +110,6 @@ pub(crate) async fn execute(
run_id,
run_spec.source_directory.as_deref(),
&run_dir,
- Arc::clone(&catalog),
)
} else {
None
@@ -237,13 +236,12 @@ fn build_fabro_run_tool_services(
current_run_id: RunId,
source_directory: Option<&str>,
run_dir: &Path,
- catalog: Arc,
) -> Option {
if worker_token.trim().is_empty() {
return None;
}
let backend = ClientBackend::new(Arc::new(client))
- .with_manifest_builder(Arc::new(WorkerRunManifestBuilder { catalog }));
+ .with_manifest_builder(Arc::new(WorkerRunManifestBuilder));
Some(FabroRunToolServices {
backend: Arc::new(backend),
current_run_id,
@@ -252,9 +250,7 @@ fn build_fabro_run_tool_services(
})
}
-struct WorkerRunManifestBuilder {
- catalog: Arc,
-}
+struct WorkerRunManifestBuilder;
impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
fn build_run_manifest(
@@ -263,12 +259,7 @@ impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> fabro_tool::ToolResult {
- run_tool_manifest::build_run_tool_manifest(
- spec,
- cwd,
- user_settings_path,
- Arc::clone(&self.catalog),
- )
+ run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}
diff --git a/lib/apps/fabro-cli/src/commands/validate.rs b/lib/apps/fabro-cli/src/commands/validate.rs
index b87460f28..e157b50b3 100644
--- a/lib/apps/fabro-cli/src/commands/validate.rs
+++ b/lib/apps/fabro-cli/src/commands/validate.rs
@@ -23,11 +23,7 @@ pub(crate) fn run(
user_settings_path: Some(active_settings_path(None)),
..Default::default()
})?;
- let response = manifest_validation::validate_manifest(
- &RunLayer::default(),
- &built.manifest,
- base_ctx.catalog()?,
- )?;
+ let response = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
let diagnostics = api_diagnostics_to_local(&response.workflow.diagnostics);
if base_ctx.json_output() {
diff --git a/lib/apps/fabro-cli/src/shared/utilities.rs b/lib/apps/fabro-cli/src/shared/utilities.rs
index e9cd32bf6..c6b048a1d 100644
--- a/lib/apps/fabro-cli/src/shared/utilities.rs
+++ b/lib/apps/fabro-cli/src/shared/utilities.rs
@@ -49,54 +49,49 @@ where
pub(crate) fn print_diagnostics(diagnostics: &[Diagnostic], styles: &Styles, printer: Printer) {
for d in diagnostics {
- let location = match (&d.node_id, &d.edge) {
- (Some(node), _) => format!(" [node: {node}]"),
- (_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
- _ => String::new(),
- };
- let source_prefix = source_prefix(d);
- match d.severity {
- Severity::Error if source_prefix.is_empty() => fabro_util::printerr!(
- printer,
- "{}{location}: {} ({})",
- styles.red.apply_to("error"),
- d.message,
- styles.dim.apply_to(&d.rule),
- ),
- Severity::Error => fabro_util::printerr!(
- printer,
- "{}: {source_prefix}{}{location} ({})",
- styles.red.apply_to("error"),
- d.message,
- styles.dim.apply_to(&d.rule),
- ),
- Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!(
- printer,
- "{}{location}: {} ({})",
- styles.yellow.apply_to("warning"),
- d.message,
- styles.dim.apply_to(&d.rule),
- ),
- Severity::Warning => fabro_util::printerr!(
- printer,
- "{}: {source_prefix}{}{location} ({})",
- styles.yellow.apply_to("warning"),
- d.message,
- styles.dim.apply_to(&d.rule),
- ),
- Severity::Info => fabro_util::printerr!(
- printer,
- "{}",
- styles.dim.apply_to(if source_prefix.is_empty() {
- format!("info{location}: {} ({})", d.message, d.rule)
- } else {
- format!("info: {source_prefix}{}{location} ({})", d.message, d.rule)
- }),
- ),
+ print_diagnostic(d, styles, printer);
+ // The fix is the actionable half of a diagnostic, so it follows every
+ // severity rather than hiding behind --verbose. Rules that have nothing
+ // useful to suggest leave it unset.
+ if let Some(fix) = &d.fix {
+ fabro_util::printerr!(printer, " {} {fix}", styles.dim.apply_to("fix:"));
}
}
}
+fn print_diagnostic(d: &Diagnostic, styles: &Styles, printer: Printer) {
+ let location = match (&d.node_id, &d.edge) {
+ (Some(node), _) => format!(" [node: {node}]"),
+ (_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
+ _ => String::new(),
+ };
+ let source_prefix = source_prefix(d);
+ let body = if source_prefix.is_empty() {
+ format!("{location}: {}", d.message)
+ } else {
+ format!(": {source_prefix}{}{location}", d.message)
+ };
+ match d.severity {
+ Severity::Error => fabro_util::printerr!(
+ printer,
+ "{}{body} ({})",
+ styles.red.apply_to("error"),
+ styles.dim.apply_to(&d.rule),
+ ),
+ Severity::Warning => fabro_util::printerr!(
+ printer,
+ "{}{body} ({})",
+ styles.yellow.apply_to("warning"),
+ styles.dim.apply_to(&d.rule),
+ ),
+ Severity::Info => fabro_util::printerr!(
+ printer,
+ "{}",
+ styles.dim.apply_to(format!("info{body} ({})", d.rule)),
+ ),
+ }
+}
+
fn source_prefix(diagnostic: &Diagnostic) -> String {
match (
diagnostic.source_path.as_deref(),
diff --git a/lib/apps/fabro-cli/tests/it/cmd/create.rs b/lib/apps/fabro-cli/tests/it/cmd/create.rs
index 5a9a4402f..488f6a1f5 100644
--- a/lib/apps/fabro-cli/tests/it/cmd/create.rs
+++ b/lib/apps/fabro-cli/tests/it/cmd/create.rs
@@ -101,6 +101,40 @@ fn create_uses_explicit_server_target_and_prints_remote_run_id() {
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
}
+#[test]
+fn create_defers_provider_validation_to_the_server() {
+ let context = test_context!();
+ let server = MockServer::start();
+ let run_id = unique_run_id();
+ let mock = server.mock(|when, then| {
+ when.method("POST")
+ .path("/api/v1/runs")
+ .body_includes(r#"provider=\"server-only\""#);
+ then.status(201)
+ .header("Content-Type", "application/json")
+ .body(run_status_response(run_id.as_str(), "submitted").to_string());
+ });
+ let output = context
+ .create_cmd()
+ .args([
+ "--server",
+ &format!("{}/api/v1", server.base_url()),
+ "--dry-run",
+ fixture("server-model.fabro").to_str().unwrap(),
+ ])
+ .output()
+ .expect("command should execute");
+
+ assert!(
+ output.status.success(),
+ "local validation should not reject a server-owned provider\nstdout:\n{}\nstderr:\n{}",
+ String::from_utf8_lossy(&output.stdout),
+ String::from_utf8_lossy(&output.stderr)
+ );
+ mock.assert();
+ assert_eq!(output_stdout(&output).trim(), run_id.as_str());
+}
+
#[test]
fn create_uses_configured_server_target_without_server_flag() {
let context = test_context!();
diff --git a/lib/apps/fabro-cli/tests/it/cmd/graph.rs b/lib/apps/fabro-cli/tests/it/cmd/graph.rs
index 6b801928a..fbf48599f 100644
--- a/lib/apps/fabro-cli/tests/it/cmd/graph.rs
+++ b/lib/apps/fabro-cli/tests/it/cmd/graph.rs
@@ -95,7 +95,9 @@ fn graph_allow_invalid_renders_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
+ fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
+ fix: Remove outgoing edges from the exit node
");
let svg = read_text(&output_path);
@@ -119,7 +121,9 @@ fn graph_invalid_workflow_fails_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
+ fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
+ fix: Remove outgoing edges from the exit node
× Validation failed
");
}
diff --git a/lib/apps/fabro-cli/tests/it/cmd/preflight.rs b/lib/apps/fabro-cli/tests/it/cmd/preflight.rs
index 8bb035f3a..4a2e44c8a 100644
--- a/lib/apps/fabro-cli/tests/it/cmd/preflight.rs
+++ b/lib/apps/fabro-cli/tests/it/cmd/preflight.rs
@@ -52,7 +52,9 @@ fn preflight_invalid_workflow_fails_with_validation_output() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
+ fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
+ fix: Remove outgoing edges from the exit node
× Validation failed
");
}
@@ -74,7 +76,9 @@ fn preflight_rejects_unbound_template_inputs() {
Goal: Demo
error: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
+ fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=`
error: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
+ fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=`
× Validation failed
");
}
diff --git a/lib/apps/fabro-cli/tests/it/cmd/validate.rs b/lib/apps/fabro-cli/tests/it/cmd/validate.rs
index 190c2817b..0741c85f1 100644
--- a/lib/apps/fabro-cli/tests/it/cmd/validate.rs
+++ b/lib/apps/fabro-cli/tests/it/cmd/validate.rs
@@ -69,6 +69,24 @@ fn simple_does_not_connect_to_configured_server() {
);
}
+/// Offline validation has no model catalog, so a model or provider the server
+/// owns must pass rather than be reported as unknown.
+#[test]
+fn server_owned_provider_is_not_rejected_by_offline_validation() {
+ let context = test_context!();
+ let mut cmd = context.validate();
+ cmd.arg(fixture("server-model.fabro"));
+ fabro_snapshot!(context.filters(), cmd, @"
+ success: true
+ exit_code: 0
+ ----- stdout -----
+ ----- stderr -----
+ Workflow: ServerModel (3 nodes, 2 edges)
+ Graph: [FIXTURES]/server-model.fabro
+ Validation: OK
+ ");
+}
+
#[test]
fn branching() {
let context = test_context!();
@@ -82,6 +100,7 @@ fn branching() {
Workflow: Branch (6 nodes, 6 edges)
Graph: [FIXTURES]/branching.fabro
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
+ fix: Add retry_target or fallback_retry_target attribute
Validation: OK
");
}
@@ -163,7 +182,9 @@ fn bare_fabro_with_unbound_inputs_validates_structurally_with_warning() {
Workflow: TemplatedUnbound (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound.fabro
warning: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
+ fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=`
warning: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
+ fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=`
Validation: OK
");
}
@@ -186,6 +207,7 @@ fn bare_fabro_with_unbound_inputs_in_imported_prompt_validates_structurally_with
Workflow: TemplatedUnboundImported (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_imported/workflow.fabro
warning: [FIXTURES]/templated_unbound_imported/work.md:1:12: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
+ fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=`
Validation: OK
");
}
@@ -207,6 +229,7 @@ fn bare_fabro_with_unbound_inputs_in_template_partial_validates_structurally_wit
Workflow: TemplatedUnboundPartial (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_partial/workflow.fabro
warning: [FIXTURES]/templated_unbound_partial/test-include.partial.md:1:4: undefined template variable `inputs.hello` in node `test_imported_include` attribute `prompt` [node: test_imported_include] (template_undefined_variable)
+ fix: bind `inputs.hello` via `[run.inputs]` in workflow.toml, or pass `--input inputs.hello=`
Validation: OK
");
}
@@ -278,6 +301,26 @@ fn validate_reports_missing_template_dependency() {
");
}
+/// A node named only by an edge is almost always a typo, so validation must
+/// fail instead of quietly running it as a default agent stage.
+#[test]
+fn edge_only_node() {
+ let context = test_context!();
+ let mut cmd = context.validate();
+ cmd.arg(fixture("edge_only_node.fabro"));
+ fabro_snapshot!(context.filters(), cmd, @"
+ success: false
+ exit_code: 1
+ ----- stdout -----
+ ----- stderr -----
+ Workflow: EdgeOnlyNode (2 nodes, 2 edges)
+ Graph: [FIXTURES]/edge_only_node.fabro
+ error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists)
+ fix: Declare node 'misspelled_node' or correct the edge endpoint
+ × Validation failed
+ ");
+}
+
#[test]
fn invalid() {
let context = test_context!();
@@ -291,7 +334,9 @@ fn invalid() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
+ fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
+ fix: Remove outgoing edges from the exit node
× Validation failed
");
}
diff --git a/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs b/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs
index 658dd1fea..d6852765e 100644
--- a/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs
+++ b/lib/apps/fabro-cli/tests/it/workflow/dry_run_examples.rs
@@ -19,6 +19,7 @@ fn dry_run_branching() {
Goal: Implement and validate a feature
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
+ fix: Add retry_target or fallback_retry_target attribute
Run: [ULID]
Web UI: http://localhost:3000/runs/[ULID]
Sandbox: local (ready in [TIME])
diff --git a/lib/apps/fabro-mcp-server/src/manifest_builder.rs b/lib/apps/fabro-mcp-server/src/manifest_builder.rs
index 53e899f1e..b09be76de 100644
--- a/lib/apps/fabro-mcp-server/src/manifest_builder.rs
+++ b/lib/apps/fabro-mcp-server/src/manifest_builder.rs
@@ -1,11 +1,8 @@
use std::path::Path;
-use std::sync::Arc;
use fabro_api::types;
-use fabro_config::load_llm_catalog_settings;
-use fabro_model::Catalog;
use fabro_server::run_tool_manifest;
-use fabro_tool::{RunManifestBuilder, ToolError, ToolResult, ValidatedCreateRunSpec};
+use fabro_tool::{RunManifestBuilder, ToolResult, ValidatedCreateRunSpec};
#[derive(Default)]
pub(crate) struct McpRunManifestBuilder;
@@ -17,20 +14,6 @@ impl RunManifestBuilder for McpRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult {
- build_mcp_run_manifest(spec, cwd, user_settings_path)
+ run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}
-
-fn build_mcp_run_manifest(
- spec: &ValidatedCreateRunSpec,
- cwd: &Path,
- user_settings_path: &Path,
-) -> ToolResult {
- let llm_catalog_settings = load_llm_catalog_settings(Some(user_settings_path))
- .map_err(|err| ToolError::message(err.to_string()))?;
- let catalog = Arc::new(
- Catalog::from_builtin_with_overrides(&llm_catalog_settings)
- .map_err(|err| ToolError::message(err.to_string()))?,
- );
- run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, catalog)
-}
diff --git a/lib/apps/fabro-server/src/demo/mod.rs b/lib/apps/fabro-server/src/demo/mod.rs
index b9d51dcef..9e4afd482 100644
--- a/lib/apps/fabro-server/src/demo/mod.rs
+++ b/lib/apps/fabro-server/src/demo/mod.rs
@@ -1134,6 +1134,7 @@ mod runs {
id: stage_id.clone(),
name: name.to_owned(),
handler,
+ billing: BilledTokenCounts::default(),
status,
wall_time_ms,
node_id: stage_id.node_id().to_owned(),
diff --git a/lib/apps/fabro-server/src/manifest_validation.rs b/lib/apps/fabro-server/src/manifest_validation.rs
index baca3f014..acc5e05e1 100644
--- a/lib/apps/fabro-server/src/manifest_validation.rs
+++ b/lib/apps/fabro-server/src/manifest_validation.rs
@@ -1,41 +1,30 @@
use std::collections::HashMap;
-use std::sync::Arc;
use anyhow::Result;
use fabro_api::types;
-use fabro_config::{EnvironmentLayer, MergeMap, RunLayer};
-use fabro_model::Catalog;
+use fabro_config::RunLayer;
use fabro_workflow::pipeline::TEMPLATE_UNDEFINED_VARIABLE_RULE;
use crate::run_manifest;
+/// Validate a manifest without a model catalog.
+///
+/// Every caller is a client — the CLI, an MCP server, a run worker — and a
+/// client's catalog is its own, not the server's. Judging model and provider
+/// availability here would reject workflows the server can run, so that is
+/// left to the server on create.
pub fn validate_manifest(
manifest_run_defaults: &RunLayer,
manifest: &types::RunManifest,
- catalog: Arc,
-) -> Result {
- validate_manifest_with_environment_defaults(
- manifest_run_defaults,
- &fabro_environment::seeded_catalog_layer(),
- manifest,
- catalog,
- )
-}
-
-pub fn validate_manifest_with_environment_defaults(
- manifest_run_defaults: &RunLayer,
- manifest_environment_defaults: &MergeMap,
- manifest: &types::RunManifest,
- catalog: Arc,
) -> Result {
let prepared = run_manifest::prepare_manifest_with_environment_defaults(
manifest_run_defaults,
- manifest_environment_defaults,
+ &fabro_environment::seeded_catalog_layer(),
&HashMap::new(),
manifest,
)?;
- let validated =
- run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?;
+ let validated = run_manifest::validate_prepared_manifest_structural(&prepared)
+ .map_err(anyhow::Error::new)?;
Ok(run_manifest::validate_response(&prepared, &validated))
}
diff --git a/lib/apps/fabro-server/src/run_manifest.rs b/lib/apps/fabro-server/src/run_manifest.rs
index 65cc6d36e..c075b9382 100644
--- a/lib/apps/fabro-server/src/run_manifest.rs
+++ b/lib/apps/fabro-server/src/run_manifest.rs
@@ -35,7 +35,8 @@ use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSecti
use fabro_validate::Severity;
use fabro_workflow::Error as WorkflowError;
use fabro_workflow::operations::{
- CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_ready_providers,
+ CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_catalog,
+ validate_with_ready_providers,
};
use fabro_workflow::pipeline::Validated;
use fabro_workflow::run_materialization::materialize_run_with_ready_providers;
@@ -192,12 +193,18 @@ pub(crate) fn validate_prepared_manifest(
validate_prepared_manifest_with_vars(prepared, catalog, HashMap::new())
}
+pub(crate) fn validate_prepared_manifest_structural(
+ prepared: &PreparedManifest,
+) -> Result {
+ validate(manifest_validate_input(prepared, HashMap::new()))
+}
+
pub(crate) fn validate_prepared_manifest_with_vars(
prepared: &PreparedManifest,
catalog: Arc,
vars: HashMap,
) -> Result {
- validate(manifest_validate_input(prepared, catalog, vars))
+ validate_with_catalog(manifest_validate_input(prepared, vars), catalog)
}
pub(crate) fn validate_prepared_manifest_for_preflight(
@@ -207,14 +214,14 @@ pub(crate) fn validate_prepared_manifest_for_preflight(
ready_providers: &[ProviderId],
) -> Result {
validate_with_ready_providers(
- manifest_validate_input(prepared, catalog, vars),
+ manifest_validate_input(prepared, vars),
+ catalog,
ready_providers,
)
}
fn manifest_validate_input(
prepared: &PreparedManifest,
- catalog: Arc,
vars: HashMap,
) -> ValidateInput {
ValidateInput {
@@ -223,7 +230,6 @@ fn manifest_validate_input(
vars,
cwd: prepared.cwd.clone(),
custom_transforms: Vec::new(),
- catalog,
}
}
diff --git a/lib/apps/fabro-server/src/run_tool_manifest.rs b/lib/apps/fabro-server/src/run_tool_manifest.rs
index 8e38e8317..584b6586f 100644
--- a/lib/apps/fabro-server/src/run_tool_manifest.rs
+++ b/lib/apps/fabro-server/src/run_tool_manifest.rs
@@ -1,20 +1,22 @@
use std::path::{Path, PathBuf};
-use std::sync::Arc;
use fabro_api::types;
use fabro_config::{CliLayer, RunGoalLayer, RunLayer};
use fabro_manifest::{ManifestBuildInput, RunOverrideInput};
-use fabro_model::Catalog;
use fabro_tool::{ToolError, ToolResult, ValidatedCreateRunSpec};
use fabro_types::settings::interp::InterpString;
use crate::manifest_validation;
+/// Build and validate a run manifest for the `fabro_run_create` tool.
+///
+/// Validation is structural. The caller is a client — an MCP server or a run
+/// worker — whose catalog is its own, not the server's, so judging model and
+/// provider availability here would reject workflows the server can run.
pub fn build_run_tool_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
- catalog: Arc,
) -> ToolResult {
let built = fabro_manifest::build_run_manifest(ManifestBuildInput {
workflow: PathBuf::from(&spec.workflow),
@@ -30,7 +32,7 @@ pub fn build_run_tool_manifest(
.map_err(|err| ToolError::from_anyhow(&err))?;
let mut validation =
- manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest, catalog)
+ manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)
.map_err(|err| ToolError::from_anyhow(&err))?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
if !validation.ok {
@@ -103,6 +105,63 @@ mod tests {
use super::*;
+ fn create_run_spec(workflow: &str) -> ValidatedCreateRunSpec {
+ ValidatedCreateRunSpec::try_from(CreateRunSpec {
+ workflow: workflow.to_string(),
+ run_id: None,
+ parent_id: None,
+ cwd: None,
+ goal: None,
+ goal_file: None,
+ inputs: HashMap::new(),
+ labels: HashMap::new(),
+ model: None,
+ provider: None,
+ environment: None,
+ dry_run: None,
+ auto_approve: None,
+ preserve_sandbox: None,
+ start: None,
+ })
+ .expect("create spec should validate")
+ }
+
+ /// The tool runs on a client, whose catalog is not the server's, so a
+ /// server-owned model must reach the server rather than fail here.
+ #[expect(
+ clippy::disallowed_methods,
+ reason = "sync test writes one workflow fixture before building the manifest"
+ )]
+ #[test]
+ fn server_owned_provider_is_not_rejected_by_tool_manifest_validation() {
+ let dir = tempfile::tempdir().expect("temp dir should be created");
+ let workflow = dir.path().join("server-model.fabro");
+ std::fs::write(
+ &workflow,
+ r#"digraph ServerModel {
+ graph [goal="Use a server-owned model"]
+ start [shape=Mdiamond]
+ work [prompt="Do work", model="private-model", provider="server-only"]
+ exit [shape=Msquare]
+ start -> work -> exit
+ }"#,
+ )
+ .expect("workflow fixture should be written");
+
+ let manifest = build_run_tool_manifest(
+ &create_run_spec(&workflow.to_string_lossy()),
+ dir.path(),
+ &dir.path().join("settings.toml"),
+ )
+ .expect("tool validation should leave provider availability to the server");
+
+ let encoded = serde_json::to_string(&manifest).expect("manifest should serialize");
+ assert!(
+ encoded.contains("server-only"),
+ "the authored provider should survive into the manifest: {encoded}"
+ );
+ }
+
#[test]
fn manifest_args_preserve_input_provenance() {
let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec {
diff --git a/lib/apps/fabro-server/src/server/handler/billing.rs b/lib/apps/fabro-server/src/server/handler/billing.rs
index a1888713c..f4fe535ee 100644
--- a/lib/apps/fabro-server/src/server/handler/billing.rs
+++ b/lib/apps/fabro-server/src/server/handler/billing.rs
@@ -2,6 +2,7 @@ use std::collections::HashMap;
use std::sync::Arc;
use chrono::{DateTime, Utc};
+use fabro_model::Catalog;
use fabro_types::{
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
};
@@ -22,6 +23,7 @@ fn run_stage_from_projection(
stage_id: &StageId,
stage: &StageProjection,
graph: &Graph,
+ catalog: &Catalog,
now: DateTime,
) -> RunStage {
let handler = stage.handler.unwrap_or_else(|| {
@@ -36,6 +38,7 @@ fn run_stage_from_projection(
id: stage_id.clone(),
name: stage_id.node_id().to_owned(),
handler,
+ billing: stage.billed_usage(Some(catalog)).into_owned(),
status: stage.effective_state(),
wall_time_ms: stage.live_wall_time_ms(now),
node_id: stage_id.node_id().to_owned(),
@@ -67,9 +70,10 @@ async fn list_run_stages(
let now = Utc::now();
let graph = projection.spec().graph();
+ let catalog = state.catalog();
let stages = projection
.iter_stages()
- .map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, now))
+ .map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, &catalog, now))
.collect::>();
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()
@@ -175,7 +179,7 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec= row.latest_visit {
@@ -188,20 +192,6 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec) -> StageTiming {
- if let Some(timing) = stage.timing {
- return timing;
- }
- if let Some(live_wall) = stage.live_wall_time_ms(now) {
- return StageTiming::wall_only(live_wall);
- }
- StageTiming::default()
-}
-
fn stage_has_billing_row(stage: &StageProjection) -> bool {
stage.completion.is_some()
|| stage.timing.is_some()
diff --git a/lib/apps/fabro-server/src/server/handler/runs.rs b/lib/apps/fabro-server/src/server/handler/runs.rs
index 0e2c8cbcb..0327d0cda 100644
--- a/lib/apps/fabro-server/src/server/handler/runs.rs
+++ b/lib/apps/fabro-server/src/server/handler/runs.rs
@@ -11,7 +11,7 @@ use axum_extra::extract::Query as ExtraQuery;
use base64::Engine as _;
use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
use bytes::Bytes;
-use chrono::Utc;
+use chrono::{DateTime, Utc};
use fabro_api::types::{
BoardColumn, RunManifest, SubmitAnswerRequest, UpdateRunParentRequest, UpdateRunRequest,
};
@@ -23,7 +23,7 @@ use fabro_store::{
};
use fabro_types::settings::ResolveError;
use fabro_types::{
- AutomationRef, Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
+ AutomationRef, Principal, Run, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
RunStatusKind, StageContextWindow, StageContextWindowStaleness,
StageContextWindowUnavailableReason, StageHandler, StageModelUsage, StageProjection,
SystemActorKind, WorkflowSettings, parse_blob_ref,
@@ -298,7 +298,7 @@ async fn validate_parent_link(
}
async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
- match state.stores.run_summaries.get(run_id, Utc::now()).await {
+ match run_summary_at(state, run_id, Utc::now()).await {
Ok(Some(summary)) => (
StatusCode::OK,
Json(state.decorate_run_summary(summary).await),
@@ -311,6 +311,28 @@ async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
}
}
+/// Read the durable summary and overlay its timing from the live projection.
+///
+/// The SQLite read model stores active timing as of the most recent event.
+/// An open inference or tool bracket keeps accruing between events, so detail
+/// reads need the projection's current estimate while the run is non-terminal.
+async fn run_summary_at(
+ state: &AppState,
+ run_id: &RunId,
+ now: DateTime,
+) -> fabro_store::Result> {
+ let Some(mut summary) = state.stores.run_summaries.get(run_id, now).await? else {
+ return Ok(None);
+ };
+ if summary.timestamps.completed_at.is_none() {
+ let projection = state.stores.runs.get_cached_projection(run_id).await?;
+ if let Some(timing) = projection.and_then(|projection| projection.live_run_timing(now)) {
+ summary.timing = Some(timing);
+ }
+ }
+ Ok(Some(summary))
+}
+
async fn list_runs(
_auth: RequiredRunManagementActor,
State(state): State>,
@@ -934,7 +956,7 @@ async fn get_run_status(
RequireRunManagementTarget(id, _actor): RequireRunManagementTarget,
State(state): State>,
) -> Response {
- match state.stores.run_summaries.get(&id, Utc::now()).await {
+ match run_summary_at(&state, &id, Utc::now()).await {
Ok(Some(run)) => {
(StatusCode::OK, Json(state.decorate_run_summary(run).await)).into_response()
}
diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs
index 71b4dfebf..83a0eeaeb 100644
--- a/lib/apps/fabro-server/src/server/tests.rs
+++ b/lib/apps/fabro-server/src/server/tests.rs
@@ -5259,6 +5259,64 @@ fn test_billed_usage(
.unwrap()
}
+async fn create_billed_retry_run(state: &Arc, run_id: RunId) {
+ create_durable_run_with_events(state, run_id, &[
+ workflow_event::Event::RunSubmitted {
+ definition_blob: None,
+ },
+ workflow_event::Event::RunStarting,
+ workflow_event::Event::RunRunning,
+ ])
+ .await;
+
+ append_scoped_stage_event(
+ state,
+ run_id,
+ "verify",
+ 1,
+ &workflow_event::Event::StageFailed {
+ node_id: "verify".to_string(),
+ name: "Verify".to_string(),
+ index: 1,
+ failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(1200),
+ billing: Some(test_billed_usage("gpt-old", 100, 10)),
+ actor: None,
+ },
+ )
+ .await;
+ append_scoped_stage_event(
+ state,
+ run_id,
+ "verify",
+ 2,
+ &workflow_event::Event::StageCompleted {
+ node_id: "verify".to_string(),
+ name: "Verify".to_string(),
+ index: 1,
+ timing: fabro_types::StageTiming::wall_only(800),
+ status: "succeeded".to_string(),
+ preferred_label: None,
+ suggested_next_ids: Vec::new(),
+ billing: Some(test_billed_usage("gpt-new", 200, 20)),
+ failure: None,
+ notes: None,
+ files_touched: Vec::new(),
+ context_updates: None,
+ jump_to_node: None,
+ context_values: None,
+ node_visits: None,
+ loop_failure_signatures: None,
+ restart_failure_signatures: None,
+ response: None,
+ attempt: 2,
+ max_attempts: 2,
+ },
+ )
+ .await;
+}
+
#[tokio::test]
async fn list_run_stages_distinguishes_visits() {
let state = test_app_state_with_isolated_storage();
@@ -5503,6 +5561,50 @@ async fn list_run_stages_exposes_execution_identity_for_resumed_stage() {
assert_eq!(second["resumed_from_stage_id"], "work@1");
}
+#[tokio::test]
+async fn run_billing_includes_live_stage_timing_in_rows_and_totals() {
+ let state = test_app_state_with_isolated_storage();
+ let app = crate::test_support::build_test_router(Arc::clone(&state));
+ let run_id = RunId::new();
+ create_durable_run_with_events(&state, run_id, &[
+ workflow_event::Event::RunSubmitted {
+ definition_blob: None,
+ },
+ workflow_event::Event::RunStarting,
+ workflow_event::Event::RunRunning,
+ workflow_run_started_event(run_id),
+ ])
+ .await;
+ append_scoped_stage_event(
+ &state,
+ run_id,
+ "work",
+ 1,
+ &stage_started_event("work", "command"),
+ )
+ .await;
+
+ tokio::time::sleep(std::time::Duration::from_millis(20)).await;
+ let response = app
+ .oneshot(
+ Request::builder()
+ .method("GET")
+ .uri(api(&format!("/runs/{run_id}/billing")))
+ .body(Body::empty())
+ .unwrap(),
+ )
+ .await
+ .unwrap();
+ let body = response_json!(response, StatusCode::OK).await;
+ let stages = body["stages"].as_array().unwrap();
+
+ assert_eq!(stages.len(), 1);
+ let row_timing = &stages[0]["timing"];
+ assert!(row_timing["active_time_ms"].as_u64().unwrap() > 0);
+ assert_eq!(row_timing["tool_time_ms"], row_timing["active_time_ms"]);
+ assert_eq!(&body["totals"]["timing"], row_timing);
+}
+
/// `checkpoint.completed_nodes` records every visit, so a looped node appears
/// once per re-entry. Billing must dedup so a retried node renders as one row
/// and `runtime_secs` is summed across all visits exactly once.
@@ -5654,65 +5756,9 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
- let failed_usage = test_billed_usage("gpt-old", 100, 10);
+
+ create_billed_retry_run(&state, run_id).await;
let success_usage = test_billed_usage("gpt-new", 200, 20);
-
- create_durable_run_with_events(&state, run_id, &[
- workflow_event::Event::RunSubmitted {
- definition_blob: None,
- },
- workflow_event::Event::RunStarting,
- workflow_event::Event::RunRunning,
- ])
- .await;
-
- append_scoped_stage_event(
- &state,
- run_id,
- "verify",
- 1,
- &workflow_event::Event::StageFailed {
- node_id: "verify".to_string(),
- name: "Verify".to_string(),
- index: 1,
- failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(1200),
- billing: Some(failed_usage),
- actor: None,
- },
- )
- .await;
- append_scoped_stage_event(
- &state,
- run_id,
- "verify",
- 2,
- &workflow_event::Event::StageCompleted {
- node_id: "verify".to_string(),
- name: "Verify".to_string(),
- index: 1,
- timing: fabro_types::StageTiming::wall_only(800),
- status: "succeeded".to_string(),
- preferred_label: None,
- suggested_next_ids: Vec::new(),
- billing: Some(success_usage.clone()),
- failure: None,
- notes: None,
- files_touched: Vec::new(),
- context_updates: None,
- jump_to_node: None,
- context_values: None,
- node_visits: None,
- loop_failure_signatures: None,
- restart_failure_signatures: None,
- response: None,
- attempt: 2,
- max_attempts: 2,
- },
- )
- .await;
-
let mut latest_outcome: Outcome> = Outcome::success();
latest_outcome.usage = Some(success_usage);
latest_outcome.timing = Some(fabro_types::StageTiming::wall_only(800));
@@ -5746,7 +5792,6 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
.unwrap();
let response = app
- .clone()
.oneshot(
Request::builder()
.method("GET")
@@ -5791,6 +5836,76 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
assert_eq!(new_model["billing"]["input_tokens"], 200);
}
+/// The stage popover reads `billing` off the stages list, so it must be scoped
+/// to one visit — unlike the Billing tab's rows, which sum every visit of a
+/// node.
+#[tokio::test]
+async fn list_run_stages_reports_billing_per_visit() {
+ let state = test_app_state_with_isolated_storage();
+ let app = crate::test_support::build_test_router(Arc::clone(&state));
+ let run_id = RunId::new();
+
+ create_billed_retry_run(&state, run_id).await;
+
+ let response = app
+ .oneshot(
+ Request::builder()
+ .method("GET")
+ .uri(api(&format!("/runs/{run_id}/stages")))
+ .body(Body::empty())
+ .unwrap(),
+ )
+ .await
+ .unwrap();
+ let body = response_json!(response, StatusCode::OK).await;
+
+ let first = stage_entry(&body, "verify@1");
+ assert_eq!(first["billing"]["input_tokens"], 100);
+ assert_eq!(first["billing"]["output_tokens"], 10);
+ assert_eq!(first["billing"]["total_usd_micros"], 110);
+
+ let second = stage_entry(&body, "verify@2");
+ assert_eq!(second["billing"]["input_tokens"], 200);
+ assert_eq!(second["billing"]["output_tokens"], 20);
+ assert_eq!(second["billing"]["total_usd_micros"], 220);
+}
+
+#[tokio::test]
+async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() {
+ let state = test_app_state_with_isolated_storage();
+ let app = crate::test_support::build_test_router(Arc::clone(&state));
+ let run_id = RunId::new();
+
+ create_durable_run_with_events(&state, run_id, &[
+ workflow_event::Event::RunSubmitted {
+ definition_blob: None,
+ },
+ workflow_event::Event::RunStarting,
+ workflow_event::Event::RunRunning,
+ ])
+ .await;
+ let started = stage_started_event("script", "command");
+ append_scoped_stage_event(&state, run_id, "script", 1, &started).await;
+
+ let response = app
+ .oneshot(
+ Request::builder()
+ .method("GET")
+ .uri(api(&format!("/runs/{run_id}/stages")))
+ .body(Body::empty())
+ .unwrap(),
+ )
+ .await
+ .unwrap();
+ let body = response_json!(response, StatusCode::OK).await;
+
+ let billing = &stage_entry(&body, "script@1")["billing"];
+ assert_eq!(billing["input_tokens"], 0);
+ assert_eq!(billing["output_tokens"], 0);
+ // No model ran, so there is nothing to price — not a $0.00 cost.
+ assert!(billing.get("total_usd_micros").is_none());
+}
+
#[tokio::test]
async fn list_run_stages_shows_retrying_after_failed_event() {
let state = test_app_state_with_isolated_storage();
@@ -7800,6 +7915,51 @@ async fn get_run_status_returns_status() {
assert!(body["labels"].is_object());
}
+#[tokio::test]
+async fn get_run_status_advances_live_active_timing_between_events() {
+ let state = test_app_state_with_isolated_storage();
+ let app = crate::test_support::build_test_router(Arc::clone(&state));
+ let run_id = RunId::new();
+ create_durable_run_with_events(&state, run_id, &[
+ workflow_event::Event::RunSubmitted {
+ definition_blob: None,
+ },
+ workflow_event::Event::RunStarting,
+ workflow_event::Event::RunRunning,
+ workflow_run_started_event(run_id),
+ ])
+ .await;
+ append_scoped_stage_event(
+ &state,
+ run_id,
+ "work",
+ 1,
+ &stage_started_event("work", "command"),
+ )
+ .await;
+
+ // The SQLite summary stores timing at the StageStarted event. A later
+ // detail read must overlay the in-flight command's active time from the
+ // projection even though no newer event has arrived.
+ tokio::time::sleep(std::time::Duration::from_millis(20)).await;
+ let response = app
+ .oneshot(
+ Request::builder()
+ .method("GET")
+ .uri(api(&format!("/runs/{run_id}")))
+ .body(Body::empty())
+ .unwrap(),
+ )
+ .await
+ .unwrap();
+ let body = response_json!(response, StatusCode::OK).await;
+ let timing = &body["timing"];
+
+ assert!(timing["active_time_ms"].as_u64().unwrap() > 0);
+ assert_eq!(timing["tool_time_ms"], timing["active_time_ms"]);
+ assert!(timing["wall_time_ms"].as_u64().unwrap() >= timing["active_time_ms"].as_u64().unwrap());
+}
+
#[tokio::test]
async fn get_run_status_not_found() {
let app = test_app_with();
diff --git a/lib/apps/fabro-server/src/test_support.rs b/lib/apps/fabro-server/src/test_support.rs
index 8fe368fff..a37c2e195 100644
--- a/lib/apps/fabro-server/src/test_support.rs
+++ b/lib/apps/fabro-server/src/test_support.rs
@@ -13,6 +13,7 @@ use axum::middleware::Next;
use axum::response::Response;
use axum::{Router, middleware};
use chrono::Duration as ChronoDuration;
+use fabro_config::user::default_storage_dir;
use fabro_config::{RunLayer, ServerSettingsBuilder, Storage, envfile};
use fabro_db::DbPool;
use fabro_interview::Interviewer;
@@ -253,9 +254,10 @@ impl TestAppStateBuilder {
self.try_build().expect("test app state should build")
}
- pub fn try_build(self) -> anyhow::Result> {
+ pub fn try_build(mut self) -> anyhow::Result> {
let (store, artifact_store) = self.store_bundle.unwrap_or_else(test_store_bundle);
let vault_path = self.vault_path.unwrap_or_else(test_secret_store_path);
+ self.server_settings = redirect_default_storage_root(self.server_settings, &vault_path);
if !self.vault_entries.is_empty() {
let mut vault = Vault::load(vault_path.clone()).expect("test vault should load");
for (name, value) in &self.vault_entries {
@@ -672,6 +674,27 @@ pub fn test_secret_store_path() -> PathBuf {
dir.join("secrets.json")
}
+/// Keeps tests off the developer's real `~/.fabro/storage`.
+///
+/// Settings built for tests usually omit `[server.storage] root`, which
+/// resolves to the production default. Handlers that walk that tree — `df`,
+/// `system/resources`, `prune` — then read whatever runs and scratch
+/// directories the machine happens to have, making tests slow and
+/// machine-dependent, and letting run-creating tests write there.
+///
+/// Only settings still carrying the production default are redirected; a test
+/// that chose its own root keeps it. The redirect goes through
+/// [`ServerSettings::with_storage_override`] so the derived local object-store
+/// roots move with it instead of pointing back at the real storage tree.
+fn redirect_default_storage_root(settings: ServerSettings, vault_path: &Path) -> ServerSettings {
+ if Path::new(&settings.server.storage.root) != default_storage_dir() {
+ return settings;
+ }
+ let root = vault_path.with_file_name("storage");
+ std::fs::create_dir_all(&root).expect("test storage root should be creatable");
+ settings.with_storage_override(&root)
+}
+
#[must_use]
pub fn test_auth_mode() -> AuthMode {
AuthMode::Enabled(ConfiguredAuth {
diff --git a/lib/components/fabro-graphviz/src/parser/semantic.rs b/lib/components/fabro-graphviz/src/parser/semantic.rs
index 8ac364cab..ed68a0fa8 100644
--- a/lib/components/fabro-graphviz/src/parser/semantic.rs
+++ b/lib/components/fabro-graphviz/src/parser/semantic.rs
@@ -1,4 +1,4 @@
-use std::collections::HashMap;
+use std::collections::{HashMap, HashSet};
use std::time::Duration;
use crate::error::Error;
@@ -57,29 +57,44 @@ fn derive_class_from_label(label: &str) -> String {
.collect()
}
+fn collect_declared_node_ids(statements: &[Statement], node_ids: &mut HashSet) {
+ for statement in statements {
+ match statement {
+ Statement::Node(node) => {
+ node_ids.insert(node.id.clone());
+ }
+ Statement::Subgraph(subgraph) => {
+ collect_declared_node_ids(&subgraph.statements, node_ids);
+ }
+ _ => {}
+ }
+ }
+}
+
struct SemanticState {
- graph: Graph,
- node_defaults: HashMap,
- edge_defaults: HashMap,
+ graph: Graph,
+ declared_node_ids: HashSet,
+ node_defaults: HashMap,
+ edge_defaults: HashMap,
}
impl SemanticState {
- fn new(name: String) -> Self {
+ fn new(name: String, declared_node_ids: HashSet) -> Self {
Self {
- graph: Graph::new(name),
+ graph: Graph::new(name),
+ declared_node_ids,
node_defaults: HashMap::new(),
edge_defaults: HashMap::new(),
}
}
- fn ensure_node(&mut self, id: &str) {
- if !self.graph.nodes.contains_key(id) {
+ fn ensure_node(&mut self, id: &str) -> &mut Node {
+ let node_defaults = &self.node_defaults;
+ self.graph.nodes.entry(id.to_string()).or_insert_with(|| {
let mut node = Node::new(id);
- for (k, v) in &self.node_defaults {
- node.attrs.insert(k.clone(), v.clone());
- }
- self.graph.nodes.insert(id.to_string(), node);
- }
+ node.attrs.clone_from(node_defaults);
+ node
+ })
}
fn add_class_to_node(node: &mut Node, cls: &str) {
@@ -90,12 +105,7 @@ impl SemanticState {
}
fn process_node(&mut self, node_stmt: &NodeStmt, subgraph_class: Option<&str>) {
- self.ensure_node(&node_stmt.id);
- let node = self
- .graph
- .nodes
- .get_mut(&node_stmt.id)
- .expect("node was just inserted by ensure_node, so get_mut cannot return None");
+ let node = self.ensure_node(&node_stmt.id);
if let Some(attrs) = &node_stmt.attrs {
for (k, v) in attrs {
node.attrs.insert(k.clone(), convert_value(v));
@@ -111,11 +121,6 @@ impl SemanticState {
.and_then(AttrValue::as_str)
.map(String::from);
if let Some(class_str) = class_str {
- let node = self
- .graph
- .nodes
- .get_mut(&node_stmt.id)
- .expect("node was just inserted by ensure_node, so get_mut cannot return None");
for cls in class_str.split(',') {
let cls = cls.trim().to_string();
if !cls.is_empty() && !node.classes.contains(&cls) {
@@ -127,12 +132,11 @@ impl SemanticState {
fn process_edge(&mut self, edge_stmt: &EdgeStmt, subgraph_class: Option<&str>) {
for id in &edge_stmt.nodes {
- self.ensure_node(id);
+ if !self.declared_node_ids.contains(id) {
+ continue;
+ }
+ let node = self.ensure_node(id);
if let Some(cls) = subgraph_class {
- let node =
- self.graph.nodes.get_mut(id).expect(
- "node was just inserted by ensure_node, so get_mut cannot return None",
- );
Self::add_class_to_node(node, cls);
}
}
@@ -251,7 +255,9 @@ impl SemanticState {
///
/// Returns an error if the AST cannot be converted to a valid graph.
pub fn ast_to_graph(dot: &DotGraph) -> Result {
- let mut state = SemanticState::new(dot.name.clone());
+ let mut declared_node_ids = HashSet::new();
+ collect_declared_node_ids(&dot.statements, &mut declared_node_ids);
+ let mut state = SemanticState::new(dot.name.clone(), declared_node_ids);
let empty = HashMap::new();
state.process_statements(&dot.statements, None, &empty, &empty);
Ok(state.graph)
@@ -515,7 +521,7 @@ mod tests {
}
#[test]
- fn ast_to_graph_implicit_nodes_from_edges() {
+ fn ast_to_graph_keeps_undeclared_edge_endpoints_out_of_nodes() {
let dot = DotGraph {
name: "Implicit".into(),
statements: vec![Statement::Edge(EdgeStmt {
@@ -524,8 +530,83 @@ mod tests {
})],
};
+ let graph = ast_to_graph(&dot).unwrap();
+ assert!(graph.nodes.is_empty());
+ assert_eq!(graph.edges, vec![Edge::new("a", "b")]);
+ }
+
+ #[test]
+ fn ast_to_graph_includes_only_declared_edge_endpoints() {
+ let dot = DotGraph {
+ name: "Declared".into(),
+ statements: vec![
+ Statement::Node(NodeStmt {
+ id: "a".into(),
+ attrs: None,
+ }),
+ Statement::Edge(EdgeStmt {
+ nodes: vec!["a".into(), "b".into()],
+ attrs: None,
+ }),
+ ],
+ };
+
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("a"));
+ assert!(!graph.nodes.contains_key("b"));
+ }
+
+ #[test]
+ fn ast_to_graph_declaration_after_edge_still_counts() {
+ let dot = DotGraph {
+ name: "DeclaredLater".into(),
+ statements: vec![
+ Statement::NodeDefaults(vec![("model".into(), AstValue::Str("first".into()))]),
+ Statement::Edge(EdgeStmt {
+ nodes: vec!["a".into(), "b".into()],
+ attrs: None,
+ }),
+ Statement::NodeDefaults(vec![("model".into(), AstValue::Str("second".into()))]),
+ Statement::Node(NodeStmt {
+ id: "b".into(),
+ attrs: Some(vec![("prompt".into(), AstValue::Str("Do it".into()))]),
+ }),
+ ],
+ };
+
+ let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("b"));
+ assert!(!graph.nodes.contains_key("a"));
+ assert_eq!(
+ graph.nodes["b"]
+ .attrs
+ .get("model")
+ .and_then(AttrValue::as_str),
+ Some("first")
+ );
+ }
+
+ #[test]
+ fn ast_to_graph_subgraph_declaration_counts() {
+ let dot = DotGraph {
+ name: "SubgraphDeclared".into(),
+ statements: vec![
+ Statement::Edge(EdgeStmt {
+ nodes: vec!["start".into(), "plan".into()],
+ attrs: None,
+ }),
+ Statement::Subgraph(SubgraphStmt {
+ name: Some("cluster_loop".into()),
+ statements: vec![Statement::Node(NodeStmt {
+ id: "plan".into(),
+ attrs: None,
+ })],
+ }),
+ ],
+ };
+
+ let graph = ast_to_graph(&dot).unwrap();
+ assert!(graph.nodes.contains_key("plan"));
+ assert!(!graph.nodes.contains_key("start"));
}
}
diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs
index 9ab1e2b2d..644048c90 100644
--- a/lib/components/fabro-store/src/run_state.rs
+++ b/lib/components/fabro-store/src/run_state.rs
@@ -19,6 +19,7 @@ use fabro_types::{
SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection,
StageModelUsage, StageOutcome, StageProjection, StageState, StartRecord, SubAgentProjection,
SubAgentStatus, TodoListKind, TodoListProjection, TodoProjection, WorkflowRef, first_event_seq,
+ timing,
};
use fabro_util::error::render_compact_with_causes;
@@ -405,7 +406,7 @@ impl RunProjectionReducer for RunProjection {
};
stage.response = response;
stage.completion = Some(completion);
- stage.timing = Some(props.timing);
+ stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@@ -428,7 +429,7 @@ impl RunProjectionReducer for RunProjection {
failure_reason,
timestamp: ts,
});
- stage.timing = Some(props.timing);
+ stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@@ -449,7 +450,7 @@ impl RunProjectionReducer for RunProjection {
context_window.event_seq = Some(event.seq);
stage.context_window = Some(context_window);
}
- close_inference_bracket(self, stored, props.visit, event.seq);
+ close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentLlmStarted(props) => {
open_inference_bracket(self, stored, props, event.seq, ts);
@@ -478,10 +479,10 @@ impl RunProjectionReducer for RunProjection {
inference.first_output_kind = None;
}
EventBody::AgentError(props) => {
- close_inference_bracket(self, stored, props.visit, event.seq);
+ close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSessionEnded(_) => {
- close_inference_brackets_for_session(self, stored);
+ close_active_brackets_for_session(self, stored, ts);
}
EventBody::AgentSessionActivated(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@@ -497,7 +498,7 @@ impl RunProjectionReducer for RunProjection {
return Ok(());
};
stage.agent_control = AgentControlState::WaitingForSteer;
- close_inference_bracket(self, stored, props.visit, event.seq);
+ close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSteeringInjected(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@@ -520,12 +521,17 @@ impl RunProjectionReducer for RunProjection {
};
stage.agent_tools.clone_from(&props.tools);
}
- // `AgentAcpStarted` is the start-of-process signal for an external
- // ACP agent. `provider_used` is intentionally sourced from the
- // subsequent `AgentSessionActivated` event, which carries the
- // canonical provider/model. ACP runs without a steering hub never
- // emit activation and so legitimately leave `provider_used`
- // unset — matching legacy ACP behavior.
+ EventBody::AgentAcpStarted(props) => {
+ let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
+ else {
+ return Ok(());
+ };
+ stage.open_acp_inference(ts);
+ // `provider_used` is intentionally sourced from the subsequent
+ // `AgentSessionActivated` event, which carries the canonical
+ // provider/model. ACP runs without a steering hub never emit
+ // activation and legitimately leave it unset.
+ }
EventBody::CommandStarted(props) => {
let script_invocation = serde_json::to_value(props).map_err(|err| {
Error::InvalidEvent(format!("invalid command.started payload: {err}"))
@@ -552,6 +558,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
+ stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@@ -564,6 +571,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
+ stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@@ -576,6 +584,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
+ stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@@ -594,6 +603,11 @@ impl RunProjectionReducer for RunProjection {
// Branches bypass the engine's StageStarted/StageCompleted
// lifecycle. Seed started_at so the branch stage drives a live
// wall-clock timer while it runs (the entry is created Running).
+ let handler = stored
+ .node_id
+ .as_deref()
+ .and_then(|node_id| self.spec().graph.nodes.get(node_id))
+ .map(|node| StageHandler::from_handler_type(node.handler_type()));
let is_new = stored
.stage_id
.as_ref()
@@ -604,6 +618,9 @@ impl RunProjectionReducer for RunProjection {
if stage.started_at.is_none() {
stage.started_at = Some(ts);
}
+ if stage.handler.is_none() {
+ stage.handler = handler;
+ }
if is_new {
stage.graph_visit = props.graph_visit;
stage
@@ -746,6 +763,11 @@ impl RunProjectionReducer for RunProjection {
});
}
EventBody::AgentToolStarted(props) => {
+ let root_session_id = if stored.parent_session_id.is_none() {
+ stored.session_id.clone()
+ } else {
+ None
+ };
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
@@ -766,6 +788,25 @@ impl RunProjectionReducer for RunProjection {
projection.invoked = true;
}
}
+ // A subagent's tools run inside the root session's tool call,
+ // so the root batch already covers them. Timing them again
+ // would double-count that span.
+ if let Some(session_id) = root_session_id {
+ stage.open_tool_call(session_id, props.tool_call_id.clone(), ts);
+ }
+ }
+ EventBody::AgentToolCompleted(props) => {
+ if stored.parent_session_id.is_some() {
+ return Ok(());
+ }
+ let Some(session_id) = stored.session_id.as_deref() else {
+ return Ok(());
+ };
+ let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
+ else {
+ return Ok(());
+ };
+ stage.close_tool_call(session_id, &props.tool_call_id, ts);
}
_ => {}
}
@@ -1031,19 +1072,19 @@ fn open_inference_bracket(
});
}
-/// Resolve the stage's `inference` slot when it holds a bracket this event is
-/// allowed to mutate.
+/// Resolve the stage owning a bracket this event is allowed to mutate,
+/// borrowing the whole projection so the caller can also fold elapsed time
+/// into the stage's live accumulators.
///
-/// `None` when the event came from a child session, when the stage has no
-/// open bracket, or when the bracket belongs to a different session — the
-/// last case matters after failover, which discards the session and builds a
-/// new one within a single visit.
-fn matching_inference_slot<'a>(
+/// Returns `None` for child-session events, for a stage with no open bracket,
+/// and for a bracket belonging to a different session (which is what
+/// post-failover events look like).
+fn matching_inference_stage<'a>(
state: &'a mut RunProjection,
stored: &RunEvent,
visit: u32,
seq: u32,
-) -> Option<&'a mut Option> {
+) -> Option<&'a mut StageProjection> {
if stored.parent_session_id.is_some() {
return None;
}
@@ -1053,7 +1094,7 @@ fn matching_inference_slot<'a>(
.inference
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
- opened_here.then_some(&mut stage.inference)
+ opened_here.then_some(stage)
}
/// Resolve the open inference bracket this event is allowed to mutate.
@@ -1063,17 +1104,39 @@ fn matching_inference_bracket<'a>(
visit: u32,
seq: u32,
) -> Option<&'a mut StageInferenceProjection> {
- matching_inference_slot(state, stored, visit, seq)?.as_mut()
+ matching_inference_stage(state, stored, visit, seq)?
+ .inference
+ .as_mut()
}
-/// Close the bracket on a stage-addressed terminal event.
-fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: u32, seq: u32) {
- if let Some(slot) = matching_inference_slot(state, stored, visit, seq) {
- *slot = None;
- }
+/// Close the bracket on a stage-addressed terminal event, folding its elapsed
+/// time into the stage's live inference accumulator.
+fn close_inference_bracket(
+ state: &mut RunProjection,
+ stored: &RunEvent,
+ visit: u32,
+ seq: u32,
+ ts: DateTime,
+) {
+ let Some(stage) = matching_inference_stage(state, stored, visit, seq) else {
+ return;
+ };
+ close_bracket_on_stage(stage, ts);
}
-/// Close every bracket opened by the session that just ended.
+/// Take the open bracket and add its span to `live_inference_ms`.
+///
+/// Retries inside the bracket are deliberately included: the in-process
+/// stopwatch counts a retried attempt's elapsed time as inference, and
+/// `agent.llm.retry` keeps the bracket open rather than reopening it.
+fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime) {
+ let Some(inference) = stage.inference.take() else {
+ return;
+ };
+ stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, ts));
+}
+
+/// Close every active bracket opened by the session that just ended.
///
/// `agent.session.ended` is the only ordering-safe backstop for terminal
/// cancel and wall-clock timeout, which tear the session down through
@@ -1088,7 +1151,11 @@ fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit:
/// opened. Implemented as a normal stage lookup it would find no target and
/// silently no-op, leaving the bracket open forever on exactly the path it
/// exists to cover.
-fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunEvent) {
+fn close_active_brackets_for_session(
+ state: &mut RunProjection,
+ stored: &RunEvent,
+ ts: DateTime,
+) {
if stored.parent_session_id.is_some() {
return;
}
@@ -1101,8 +1168,9 @@ fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunE
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
if opened_here {
- stage.inference = None;
+ close_bracket_on_stage(stage, ts);
}
+ stage.close_tool_batch_for_session(session_id, ts);
}
}
@@ -1412,23 +1480,26 @@ fn finalize_unfinished_stages_after_run_failed(
StageState::Failed
};
- for (_, stage) in state.iter_stages_mut() {
+ for (_, stage) in state.iter_stages_unordered_mut() {
if stage.state.is_terminal() {
continue;
}
+ // Close any brackets still open so their spans are not dropped on the
+ // floor when the live estimate is frozen into `timing` below.
+ close_bracket_on_stage(stage, timestamp);
+ stage.close_open_acp_inference(timestamp);
+ stage.close_open_tool_batch(timestamp);
+
+ // Freeze the live estimate before flipping to a terminal state:
+ // `live_timing` reads `effective_state` and would return wall-only
+ // once the stage no longer looks in-flight.
+ let frozen = stage.live_timing(timestamp);
stage.state = terminal_state;
- if stage.timing.is_none() {
- if let Some(started_at) = stage.started_at {
- let wall_time_ms = u64::try_from(
- timestamp
- .signed_duration_since(started_at)
- .num_milliseconds()
- .max(0),
- )
- .expect("non-negative milliseconds fit in u64");
- stage.timing = Some(fabro_types::StageTiming::wall_only(wall_time_ms));
- }
+ if stage.timing.is_none() && stage.started_at.is_some() {
+ stage.set_authoritative_timing(frozen);
+ } else {
+ stage.clear_live_timing();
}
}
}
@@ -1545,21 +1616,519 @@ mod tests {
};
use fabro_types::settings::run::{DockerfileSource, EnvironmentProvider};
use fabro_types::{
- AgentBackend, AgentControlState, AutomationRef, BilledModelUsage, BilledTokenCounts,
- BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody,
- FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Outcome,
- PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
+ AgentBackend, AgentControlState, AttrValue, AutomationRef, BilledModelUsage,
+ BilledTokenCounts, BlockedReason, Checkpoint, CheckpointRecord, CommandTermination,
+ EventBody, FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Node,
+ Outcome, PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
RunApprovalState, RunBlobId, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec,
RunStatus, Speed, StageContextWindowBreakdownItem, StageContextWindowCategory,
StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
- StageContextWindowWarning, StageModelUsage, StageOutcome, StageState, SubAgentStatus,
- SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support,
+ StageContextWindowWarning, StageHandler, StageModelUsage, StageOutcome, StageState,
+ StageTiming, SubAgentStatus, SuccessReason, WorkflowSettings, first_event_seq, fixtures,
+ test_support,
};
use serde_json::json;
use super::{RunProjection, RunProjectionReducer, build_summary};
use crate::{Error, EventEnvelope, StageId};
+ /// Live accumulation of inference and tool time while a stage is in
+ /// flight. The finalized breakdown still arrives with the terminal event
+ /// and replaces these; these exist so a long-running stage is not reported
+ /// as doing no work.
+ mod live_active_accumulation {
+ use fabro_types::run_event::{
+ AgentLlmFirstOutputProps, AgentLlmRetryProps, AgentLlmStartedProps,
+ AgentToolCompletedProps, AgentToolStartedProps,
+ };
+ use fabro_types::{
+ LlmOutputKind, LlmRetryPhase, ModelRef, Speed, StageOutcome, StageProjection,
+ };
+
+ use super::*;
+
+ fn stage_id() -> StageId {
+ StageId::new("plan", 1)
+ }
+
+ fn session_event(seq: u32, ts: &str, session_id: &str, body: EventBody) -> EventEnvelope {
+ let mut event = test_stage_event_at(seq, ts, body, stage_id());
+ event.event.session_id = Some(session_id.to_string());
+ event
+ }
+
+ fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
+ session_event(seq, ts, "session-1", body)
+ }
+
+ /// An event from a sub-agent session nested under the root session.
+ fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
+ let mut event = agent_event(seq, ts, body);
+ event.event.session_id = Some("session-child".to_string());
+ event.event.parent_session_id = Some("session-1".to_string());
+ event
+ }
+
+ fn llm_started() -> EventBody {
+ EventBody::AgentLlmStarted(AgentLlmStartedProps {
+ requested_model: ModelRef {
+ provider: "anthropic".parse().unwrap(),
+ model_id: "claude-fable-5".into(),
+ speed: Some(Speed::Fast),
+ },
+ visit: 1,
+ })
+ }
+
+ fn tool_started(tool_call_id: &str) -> EventBody {
+ EventBody::AgentToolStarted(AgentToolStartedProps {
+ tool_name: "Bash".to_string(),
+ tool_call_id: tool_call_id.to_string(),
+ arguments: json!({}),
+ visit: 1,
+ tool_call: None,
+ turn_id: None,
+ parent_message_id: None,
+ })
+ }
+
+ fn tool_completed(tool_call_id: &str) -> EventBody {
+ EventBody::AgentToolCompleted(AgentToolCompletedProps {
+ tool_name: "Bash".to_string(),
+ tool_call_id: tool_call_id.to_string(),
+ output: json!("ok"),
+ is_error: false,
+ visit: 1,
+ tool_result: None,
+ turn_id: None,
+ })
+ }
+
+ fn agent_message() -> EventBody {
+ EventBody::AgentMessage(live_agent_message_props(live_counts(10, 5)))
+ }
+
+ fn started_state() -> RunProjection {
+ let mut state = initialized_projection();
+ state
+ .apply_event(&test_stage_event_at(
+ 1,
+ "2026-04-07T12:00:00Z",
+ EventBody::StageStarted(started_props()),
+ stage_id(),
+ ))
+ .unwrap();
+ state
+ }
+
+ fn stage(state: &RunProjection) -> &StageProjection {
+ state.stage(&stage_id()).unwrap()
+ }
+
+ #[test]
+ fn closing_an_inference_bracket_accumulates_its_span() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 3,
+ "2026-04-07T12:00:06Z",
+ EventBody::AgentLlmFirstOutput(AgentLlmFirstOutputProps {
+ kind: LlmOutputKind::Text,
+ visit: 1,
+ }),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(4, "2026-04-07T12:00:12Z", agent_message()))
+ .unwrap();
+
+ // 12:00:05 -> 12:00:12; first_output is a marker, not the close.
+ assert_eq!(stage(&state).live_inference_ms, 7_000);
+ assert!(stage(&state).inference.is_none());
+ }
+
+ #[test]
+ fn concurrent_tool_calls_count_once_not_per_call() {
+ let mut state = started_state();
+ for (seq, id) in [(2, "call-a"), (3, "call-b"), (4, "call-c")] {
+ state
+ .apply_event(&agent_event(seq, "2026-04-07T12:00:00Z", tool_started(id)))
+ .unwrap();
+ }
+ // All three finish 10s later. Summing per-call spans would report
+ // 30s; the batch actually occupied 10s of wall time.
+ for (seq, id) in [(5, "call-a"), (6, "call-b"), (7, "call-c")] {
+ state
+ .apply_event(&agent_event(
+ seq,
+ "2026-04-07T12:00:10Z",
+ tool_completed(id),
+ ))
+ .unwrap();
+ }
+
+ assert_eq!(stage(&state).live_tool_ms, 10_000);
+ assert!(stage(&state).tool_batch.is_none());
+ }
+
+ #[test]
+ fn a_batch_stays_open_until_its_last_call_reports() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(
+ 2,
+ "2026-04-07T12:00:00Z",
+ tool_started("call-a"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 3,
+ "2026-04-07T12:00:02Z",
+ tool_started("call-b"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 4,
+ "2026-04-07T12:00:05Z",
+ tool_completed("call-a"),
+ ))
+ .unwrap();
+
+ assert_eq!(
+ stage(&state).live_tool_ms,
+ 0,
+ "batch must not close while call-b is outstanding"
+ );
+
+ state
+ .apply_event(&agent_event(
+ 5,
+ "2026-04-07T12:00:09Z",
+ tool_completed("call-b"),
+ ))
+ .unwrap();
+
+ // Measured from the batch open, not from the last call's start.
+ assert_eq!(stage(&state).live_tool_ms, 9_000);
+ }
+
+ #[test]
+ fn successive_batches_accumulate() {
+ let mut state = started_state();
+ for (seq, ts, body) in [
+ (2, "2026-04-07T12:00:00Z", tool_started("call-a")),
+ (3, "2026-04-07T12:00:04Z", tool_completed("call-a")),
+ (4, "2026-04-07T12:00:10Z", tool_started("call-b")),
+ (5, "2026-04-07T12:00:16Z", tool_completed("call-b")),
+ ] {
+ state.apply_event(&agent_event(seq, ts, body)).unwrap();
+ }
+
+ assert_eq!(stage(&state).live_tool_ms, 10_000);
+ }
+
+ #[test]
+ fn a_duplicate_completion_does_not_drain_the_batch_early() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(
+ 2,
+ "2026-04-07T12:00:00Z",
+ tool_started("call-a"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 3,
+ "2026-04-07T12:00:00Z",
+ tool_started("call-b"),
+ ))
+ .unwrap();
+ // call-a reports twice, as a replayed or duplicated log can.
+ state
+ .apply_event(&agent_event(
+ 4,
+ "2026-04-07T12:00:03Z",
+ tool_completed("call-a"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 5,
+ "2026-04-07T12:00:04Z",
+ tool_completed("call-a"),
+ ))
+ .unwrap();
+
+ assert_eq!(stage(&state).live_tool_ms, 0);
+ assert!(stage(&state).tool_batch.is_some());
+ }
+
+ #[test]
+ fn a_foreign_session_completion_does_not_mutate_the_open_batch() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(
+ 2,
+ "2026-04-07T12:00:00Z",
+ tool_started("call-a"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&session_event(
+ 3,
+ "2026-04-07T12:00:05Z",
+ "session-2",
+ tool_completed("call-a"),
+ ))
+ .unwrap();
+
+ let batch = stage(&state).tool_batch.as_ref().unwrap();
+ assert_eq!(batch.session_id, "session-1");
+ assert!(batch.open_call_ids.contains("call-a"));
+ assert_eq!(stage(&state).live_tool_ms, 0);
+ }
+
+ #[test]
+ fn a_replacement_session_starts_a_separate_tool_batch() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(
+ 2,
+ "2026-04-07T12:00:00Z",
+ tool_started("old-call"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&session_event(
+ 3,
+ "2026-04-07T12:00:05Z",
+ "session-2",
+ tool_started("new-call"),
+ ))
+ .unwrap();
+
+ assert_eq!(stage(&state).live_tool_ms, 5_000);
+ let batch = stage(&state).tool_batch.as_ref().unwrap();
+ assert_eq!(batch.session_id, "session-2");
+ assert_eq!(
+ batch.open_call_ids,
+ ["new-call".to_string()].into_iter().collect()
+ );
+
+ // A delayed completion from the old session cannot close the new
+ // session's batch even when call ids happen to collide.
+ state
+ .apply_event(&agent_event(
+ 4,
+ "2026-04-07T12:00:07Z",
+ tool_completed("new-call"),
+ ))
+ .unwrap();
+ assert!(stage(&state).tool_batch.is_some());
+
+ state
+ .apply_event(&session_event(
+ 5,
+ "2026-04-07T12:00:09Z",
+ "session-2",
+ tool_completed("new-call"),
+ ))
+ .unwrap();
+ assert_eq!(stage(&state).live_tool_ms, 9_000);
+ assert!(stage(&state).tool_batch.is_none());
+ }
+
+ #[test]
+ fn subagent_tool_calls_do_not_double_count_against_the_root_batch() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(
+ 2,
+ "2026-04-07T12:00:00Z",
+ tool_started("root-call"),
+ ))
+ .unwrap();
+ // The sub-agent's own tools run inside the root call's span.
+ state
+ .apply_event(&child_event(
+ 3,
+ "2026-04-07T12:00:01Z",
+ tool_started("child-call"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&child_event(
+ 4,
+ "2026-04-07T12:00:02Z",
+ tool_completed("child-call"),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 5,
+ "2026-04-07T12:00:08Z",
+ tool_completed("root-call"),
+ ))
+ .unwrap();
+
+ assert_eq!(stage(&state).live_tool_ms, 8_000);
+ }
+
+ #[test]
+ fn session_end_accumulates_every_open_active_bracket() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 3,
+ "2026-04-07T12:00:07Z",
+ tool_started("call-a"),
+ ))
+ .unwrap();
+
+ let mut ended = test_stage_event_at(
+ 4,
+ "2026-04-07T12:00:20Z",
+ EventBody::AgentSessionEnded(AgentSessionEndedProps {}),
+ stage_id(),
+ );
+ ended.event.session_id = Some("session-1".to_string());
+ state.apply_event(&ended).unwrap();
+
+ assert_eq!(stage(&state).live_inference_ms, 15_000);
+ assert_eq!(stage(&state).live_tool_ms, 13_000);
+ assert!(stage(&state).inference.is_none());
+ assert!(stage(&state).tool_batch.is_none());
+ }
+
+ #[test]
+ fn a_foreign_session_close_leaves_the_bracket_and_accumulator_alone() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
+ .unwrap();
+
+ // Post-failover: a new session emits the message, so the old
+ // bracket is not this event's to close or bill.
+ let mut foreign =
+ test_stage_event_at(3, "2026-04-07T12:00:20Z", agent_message(), stage_id());
+ foreign.event.session_id = Some("session-2".to_string());
+ state.apply_event(&foreign).unwrap();
+
+ assert_eq!(stage(&state).live_inference_ms, 0);
+ assert!(stage(&state).inference.is_some());
+ }
+
+ #[test]
+ fn a_retry_keeps_accumulating_within_one_bracket() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 3,
+ "2026-04-07T12:00:04Z",
+ EventBody::AgentLlmRetry(AgentLlmRetryProps {
+ provider: "anthropic".to_string(),
+ model: "claude-fable-5".to_string(),
+ attempt: 0,
+ delay_secs: 0.0,
+ error: json!({ "kind": "stream" }),
+ phase: Some(LlmRetryPhase::Consume),
+ visit: 1,
+ }),
+ ))
+ .unwrap();
+ state
+ .apply_event(&agent_event(4, "2026-04-07T12:00:11Z", agent_message()))
+ .unwrap();
+
+ // The whole bracket counts, retry included, matching the
+ // in-process stopwatch.
+ assert_eq!(stage(&state).live_inference_ms, 11_000);
+ assert_eq!(stage(&state).inference, None);
+ }
+
+ #[test]
+ fn stage_completion_replaces_the_live_estimate_with_finalized_timing() {
+ let mut state = started_state();
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(3, "2026-04-07T12:00:09Z", agent_message()))
+ .unwrap();
+ assert_eq!(stage(&state).live_inference_ms, 9_000);
+
+ state
+ .apply_event(&test_stage_event_at(
+ 4,
+ "2026-04-07T12:00:10Z",
+ EventBody::StageCompleted(completed_props(10_000, StageOutcome::Succeeded)),
+ stage_id(),
+ ))
+ .unwrap();
+
+ let stage = stage(&state);
+ assert_eq!(
+ stage.live_timing(test_dt("2026-04-07T12:30:00Z")),
+ stage.timing.unwrap(),
+ "a terminal stage reports its finalized breakdown, not a live estimate"
+ );
+ assert_eq!(stage.live_inference_ms, 0);
+ assert_eq!(stage.live_tool_ms, 0);
+ assert!(stage.inference.is_none());
+ assert!(stage.tool_batch.is_none());
+ }
+
+ #[test]
+ fn run_failure_freezes_open_work_and_clears_live_bookkeeping() {
+ let mut state = started_state();
+ state.status = RunStatus::Running;
+ state
+ .apply_event(&agent_event(2, "2026-04-07T12:00:01Z", llm_started()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(3, "2026-04-07T12:00:04Z", agent_message()))
+ .unwrap();
+ state
+ .apply_event(&agent_event(
+ 4,
+ "2026-04-07T12:00:05Z",
+ tool_started("call-a"),
+ ))
+ .unwrap();
+
+ let mut failed = test_event(
+ 5,
+ EventBody::RunFailed(run_failed_props(FailureReason::WorkflowError)),
+ None,
+ );
+ failed.event.ts = test_dt("2026-04-07T12:00:10Z");
+ state.apply_event(&failed).unwrap();
+
+ let stage = stage(&state);
+ assert_eq!(
+ stage.timing,
+ Some(fabro_types::StageTiming::new(10_000, 3_000, 5_000))
+ );
+ assert_eq!(stage.state, StageState::Failed);
+ assert_eq!(stage.live_inference_ms, 0);
+ assert_eq!(stage.live_tool_ms, 0);
+ assert!(stage.inference.is_none());
+ assert!(stage.tool_batch.is_none());
+ }
+ }
+
fn test_event(seq: u32, body: EventBody, node_id: Option<&str>) -> EventEnvelope {
let event = RunEvent {
id: format!("evt-{seq}"),
@@ -2365,6 +2934,49 @@ mod tests {
);
}
+ #[test]
+ fn parallel_branch_started_uses_the_graph_handler_for_live_timing() {
+ for (handler_type, handler, expected) in [
+ (
+ "prompt",
+ StageHandler::Prompt,
+ StageTiming::new(5_000, 5_000, 0),
+ ),
+ (
+ "command",
+ StageHandler::Command,
+ StageTiming::new(5_000, 0, 5_000),
+ ),
+ ] {
+ let mut spec = test_run_spec();
+ let mut node = Node::new("review");
+ node.attrs.insert(
+ "type".to_string(),
+ AttrValue::String(handler_type.to_string()),
+ );
+ spec.graph.nodes.insert(node.id.clone(), node);
+ let mut state = RunProjection::new("Test run".to_string(), spec, Utc::now());
+ let branch = StageId::new("review", 1);
+
+ state
+ .apply_event(&test_stage_event_at(
+ 3,
+ "2026-04-07T12:00:00Z",
+ EventBody::ParallelBranchStarted(ParallelBranchStartedProps {
+ index: 0,
+ graph_visit: None,
+ resumed_from_stage_id: None,
+ }),
+ branch.clone(),
+ ))
+ .unwrap();
+
+ let stage = state.stage(&branch).unwrap();
+ assert_eq!(stage.handler, Some(handler));
+ assert_eq!(stage.live_timing(test_dt("2026-04-07T12:00:05Z")), expected);
+ }
+ }
+
fn start_stage(state: &mut RunProjection, stage_id: &StageId) {
state
.apply_event(&test_stage_event(
@@ -2509,6 +3121,66 @@ mod tests {
assert_eq!(provider_used.model.as_deref(), Some("fake"));
}
+ #[test]
+ fn acp_events_accumulate_live_inference_time() {
+ let mut state = initialized_projection();
+ let stage_id = StageId::new("code", 1);
+
+ state
+ .apply_event(&test_stage_event_at(
+ 3,
+ "2026-04-07T12:00:00Z",
+ EventBody::StageStarted(StageStartedProps {
+ graph_visit: None,
+ resumed_from_stage_id: None,
+ index: 0,
+ handler_type: "agent".to_string(),
+ attempt: 1,
+ max_attempts: 1,
+ }),
+ stage_id.clone(),
+ ))
+ .unwrap();
+ state
+ .apply_event(&test_stage_event_at(
+ 4,
+ "2026-04-07T12:00:05Z",
+ EventBody::AgentAcpStarted(AgentAcpStartedProps {
+ visit: 1,
+ command: "python fake_agent.py".to_string(),
+ config_name: Some("fake".to_string()),
+ }),
+ stage_id.clone(),
+ ))
+ .unwrap();
+
+ let stage = state.stage(&stage_id).unwrap();
+ assert_eq!(
+ stage.live_timing(test_dt("2026-04-07T12:00:15Z")),
+ StageTiming::new(15_000, 10_000, 0)
+ );
+
+ state
+ .apply_event(&test_stage_event_at(
+ 5,
+ "2026-04-07T12:00:17Z",
+ EventBody::AgentAcpCompleted(AgentAcpCompletedProps {
+ stdout: "done".to_string(),
+ stderr: String::new(),
+ stop_reason: "end_turn".to_string(),
+ duration_ms: 12_000,
+ }),
+ stage_id.clone(),
+ ))
+ .unwrap();
+
+ let stage = state.stage(&stage_id).unwrap();
+ assert_eq!(
+ stage.live_timing(test_dt("2026-04-07T12:00:20Z")),
+ StageTiming::new(20_000, 12_000, 0)
+ );
+ }
+
#[test]
fn agent_acp_completed_updates_stage_output_projection() {
let mut state = initialized_projection();
diff --git a/lib/components/fabro-store/src/run_summary_store.rs b/lib/components/fabro-store/src/run_summary_store.rs
index 42e1d72a7..d80936dc7 100644
--- a/lib/components/fabro-store/src/run_summary_store.rs
+++ b/lib/components/fabro-store/src/run_summary_store.rs
@@ -3,7 +3,7 @@ use std::fmt::Write as _;
use std::sync::LazyLock;
use chrono::{DateTime, Utc};
-use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming};
+use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming, timing};
use sqlx::sqlite::{SqliteConnection, SqliteRow};
use sqlx::{QueryBuilder, Row as _, Sqlite, SqlitePool};
use strum::VariantArray as _;
@@ -545,7 +545,7 @@ fn overlay_live_wall_time(run: &mut Run, now: DateTime) {
let Some(started_at) = run.timestamps.started_at else {
return;
};
- let wall_time_ms = RunTiming::wall_time_ms_since(started_at, now);
+ let wall_time_ms = timing::elapsed_ms(started_at, now);
run.timing = Some(
run.timing
.unwrap_or_else(|| RunTiming::wall_only(wall_time_ms))
diff --git a/lib/components/fabro-store/src/slate/mod.rs b/lib/components/fabro-store/src/slate/mod.rs
index 617d56565..275084dbb 100644
--- a/lib/components/fabro-store/src/slate/mod.rs
+++ b/lib/components/fabro-store/src/slate/mod.rs
@@ -327,6 +327,18 @@ impl Database {
Ok(self.projection_cache.get(run_id).await)
}
+ pub async fn get_cached_projection(
+ &self,
+ run_id: &RunId,
+ ) -> Result>> {
+ self.warm_projection_cache().await?;
+ Ok(self
+ .projection_cache
+ .projection_snapshot(run_id)
+ .await
+ .map(|(projection, _)| projection))
+ }
+
pub async fn get_cached_summary(
&self,
run_id: &RunId,
diff --git a/lib/components/fabro-validate/src/rules/edge_target_exists.rs b/lib/components/fabro-validate/src/rules/edge_target_exists.rs
index 8cfe67282..3cfbe2130 100644
--- a/lib/components/fabro-validate/src/rules/edge_target_exists.rs
+++ b/lib/components/fabro-validate/src/rules/edge_target_exists.rs
@@ -1,3 +1,5 @@
+use std::collections::HashSet;
+
use fabro_graphviz::graph::Graph;
use crate::{Diagnostic, LintRule, Severity};
@@ -8,6 +10,26 @@ pub(super) fn rule() -> Box {
struct Rule;
+impl Rule {
+ fn diagnostic(&self, node_id: &str, from: &str, to: &str) -> Diagnostic {
+ Diagnostic {
+ rule: self.name().to_string(),
+ severity: Severity::Error,
+ message: format!(
+ "Node '{node_id}' is referenced by edge '{from} -> {to}' but has no node \
+ declaration"
+ ),
+ node_id: Some(node_id.to_string()),
+ edge: Some((from.to_string(), to.to_string())),
+ fix: Some(format!(
+ "Declare node '{node_id}' or correct the edge endpoint"
+ )),
+
+ ..Diagnostic::default()
+ }
+ }
+}
+
impl LintRule for Rule {
fn name(&self) -> &'static str {
"edge_target_exists"
@@ -15,36 +37,12 @@ impl LintRule for Rule {
fn apply(&self, graph: &Graph) -> Vec {
let mut diagnostics = Vec::new();
+ let mut reported = HashSet::new();
for edge in &graph.edges {
- if !graph.nodes.contains_key(&edge.to) {
- diagnostics.push(Diagnostic {
- rule: self.name().to_string(),
- severity: Severity::Error,
- message: format!(
- "Edge from '{}' targets non-existent node '{}'",
- edge.from, edge.to
- ),
- node_id: None,
- edge: Some((edge.from.clone(), edge.to.clone())),
- fix: Some(format!("Define node '{}' or fix the edge target", edge.to)),
-
- ..Diagnostic::default()
- });
- }
- if !graph.nodes.contains_key(&edge.from) {
- diagnostics.push(Diagnostic {
- rule: self.name().to_string(),
- severity: Severity::Error,
- message: format!("Edge source '{}' references non-existent node", edge.from),
- node_id: None,
- edge: Some((edge.from.clone(), edge.to.clone())),
- fix: Some(format!(
- "Define node '{}' or fix the edge source",
- edge.from
- )),
-
- ..Diagnostic::default()
- });
+ for endpoint in [&edge.to, &edge.from] {
+ if !graph.nodes.contains_key(endpoint) && reported.insert(endpoint) {
+ diagnostics.push(self.diagnostic(endpoint, &edge.from, &edge.to));
+ }
}
}
diagnostics
@@ -53,11 +51,112 @@ impl LintRule for Rule {
#[cfg(test)]
mod tests {
- use fabro_graphviz::graph::Edge;
+ use fabro_graphviz::graph::{Edge, Graph};
+ use fabro_graphviz::parser;
use super::Rule;
use crate::rules::test_support::minimal_graph;
- use crate::{LintRule, Severity};
+ use crate::{Diagnostic, LintRule, Severity};
+
+ fn parse(dot: &str) -> Graph {
+ parser::parse(dot).expect("fixture should parse")
+ }
+
+ fn undeclared_nodes(graph: &Graph) -> Vec {
+ Rule.apply(graph)
+ .into_iter()
+ .map(|d| d.node_id.expect("diagnostic should name a node"))
+ .collect()
+ }
+
+ #[test]
+ fn edge_only_node_is_rejected() {
+ let graph = parse(
+ r"digraph EdgeOnly {
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+ start -> misspelled_node
+ misspelled_node -> exit
+ }",
+ );
+
+ let diagnostics = Rule.apply(&graph);
+ assert_eq!(diagnostics.len(), 1, "diagnostics: {diagnostics:?}");
+ let Diagnostic {
+ severity,
+ node_id,
+ edge,
+ ..
+ } = &diagnostics[0];
+ assert_eq!(*severity, Severity::Error);
+ assert_eq!(node_id.as_deref(), Some("misspelled_node"));
+ assert_eq!(
+ edge.clone(),
+ Some(("start".to_string(), "misspelled_node".to_string()))
+ );
+ }
+
+ #[test]
+ fn declaration_after_the_edge_is_accepted() {
+ let graph = parse(
+ r#"digraph DeclaredLater {
+ start -> work
+ work [prompt="Do the work"]
+ work -> exit
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+ }"#,
+ );
+
+ assert!(Rule.apply(&graph).is_empty());
+ }
+
+ #[test]
+ fn chained_edges_report_every_undeclared_endpoint() {
+ let graph = parse(
+ r"digraph Chained {
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+ start -> first -> second -> exit
+ }",
+ );
+
+ assert_eq!(undeclared_nodes(&graph), vec!["first", "second"]);
+ }
+
+ #[test]
+ fn a_node_is_reported_once_no_matter_how_many_edges_use_it() {
+ let graph = parse(
+ r"digraph Repeated {
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+ start -> typo
+ typo -> exit
+ typo -> start
+ }",
+ );
+
+ assert_eq!(undeclared_nodes(&graph), vec!["typo"]);
+ }
+
+ #[test]
+ fn subgraph_declaration_is_accepted() {
+ let graph = parse(
+ r#"digraph Subgraphed {
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+
+ subgraph cluster_loop {
+ label = "Loop A"
+ plan [prompt="Plan the work"]
+ }
+
+ start -> plan -> exit
+ }"#,
+ );
+
+ assert!(Rule.apply(&graph).is_empty());
+ }
#[test]
fn edge_target_exists_rule_missing_target() {
diff --git a/lib/components/fabro-workflow/src/billing_rollup.rs b/lib/components/fabro-workflow/src/billing_rollup.rs
index 4ba8976bc..09256429f 100644
--- a/lib/components/fabro-workflow/src/billing_rollup.rs
+++ b/lib/components/fabro-workflow/src/billing_rollup.rs
@@ -1,32 +1,7 @@
-use std::borrow::Cow;
use std::collections::HashMap;
use fabro_model::Catalog;
-use fabro_types::{
- BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageTiming,
-};
-
-fn stage_usage_with_cost<'a>(
- catalog: Option<&Catalog>,
- stage: &'a StageProjection,
-) -> Cow<'a, BilledTokenCounts> {
- let Some(catalog) = catalog else {
- return Cow::Borrowed(&stage.usage);
- };
- let Some(model) = stage.model.as_ref() else {
- return Cow::Borrowed(&stage.usage);
- };
- if stage.usage.total_usd_micros.is_some() {
- return Cow::Borrowed(&stage.usage);
- }
-
- let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
- return Cow::Borrowed(&stage.usage);
- };
- let mut usage = stage.usage.clone();
- usage.total_usd_micros = Some(total_usd_micros);
- Cow::Owned(usage)
-}
+use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageTiming};
#[derive(Debug, Clone, PartialEq)]
pub struct ProjectionBillingStage {
@@ -80,7 +55,7 @@ pub fn billing_rollup_from_projection(
if is_boundary_stage(projection, stage_id.node_id()) {
continue;
}
- let usage = stage_usage_with_cost(catalog, stage);
+ let usage = stage.billed_usage(catalog);
let usage = usage.as_ref();
if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() {
continue;
diff --git a/lib/components/fabro-workflow/src/handler/manager_loop.rs b/lib/components/fabro-workflow/src/handler/manager_loop.rs
index 98804d5e9..5668954c6 100644
--- a/lib/components/fabro-workflow/src/handler/manager_loop.rs
+++ b/lib/components/fabro-workflow/src/handler/manager_loop.rs
@@ -16,7 +16,7 @@ use crate::artifact_upload::ArtifactSink;
use crate::condition::evaluate_condition;
use crate::context::{Context, WorkflowContext, context_diff_public, keys};
use crate::error::Error;
-use crate::operations::{ValidateInput, WorkflowInput, validate};
+use crate::operations::{ValidateInput, WorkflowInput, validate_with_catalog};
use crate::outcome::{Outcome, OutcomeExt, StageOutcome};
use crate::pipeline::types::Initialized;
use crate::run_options::RunOptions;
@@ -65,20 +65,14 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result Result Some(workflow.path.clone()),
WorkflowInput::Path(_) | WorkflowInput::DotSource { .. } => None,
};
- let mut validated = validate(ValidateInput {
- workflow,
- settings: WorkflowSettings::default(),
- vars: std::collections::HashMap::new(),
- cwd,
- custom_transforms: Vec::new(),
- catalog: Arc::clone(&services.run.catalog),
- })?;
- validated.promote_template_undefined_variables_to_errors();
- validated.raise_on_errors()?;
- let (graph, _, _) = validated.into_parts();
+ let graph = validate_child_workflow(workflow, cwd, services)?;
return Ok(ParsedChildWorkflow {
graph,
workflow_path,
@@ -132,6 +116,29 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result Result {
+ let mut validated = validate_with_catalog(
+ ValidateInput {
+ workflow,
+ settings: WorkflowSettings::default(),
+ vars: HashMap::new(),
+ cwd,
+ custom_transforms: Vec::new(),
+ },
+ Arc::clone(&services.run.catalog),
+ )?;
+ validated.promote_template_undefined_variables_to_errors();
+ validated.raise_on_errors()?;
+ let (graph, _, _) = validated.into_parts();
+ Ok(graph)
+}
+
#[async_trait]
impl Handler for SubWorkflowHandler {
async fn execute(
diff --git a/lib/components/fabro-workflow/src/operations/create.rs b/lib/components/fabro-workflow/src/operations/create.rs
index 2ec40d443..4f810cb54 100644
--- a/lib/components/fabro-workflow/src/operations/create.rs
+++ b/lib/components/fabro-workflow/src/operations/create.rs
@@ -28,7 +28,7 @@ use crate::pipeline::{self, Persisted, TransformOptions, Validated};
use crate::records::RunSpec;
use crate::run_lookup::default_scratch_base;
use crate::run_materialization::materialize_run;
-use crate::transforms::{RenderMode, Transform};
+use crate::transforms::{ModelResolutionTransform, RenderMode};
use crate::workflow_bundle::{RunDefinition, WorkflowBundle};
#[derive(Clone, Debug)]
@@ -293,28 +293,21 @@ fn create_from_source(
file_resolver: Option>,
goal_override: Option<&str>,
) -> Result {
- let template_context = template_context(Some(&options.settings), vars);
- let mut validated = preprocess_and_validate(
- dot_source,
- options.source_name.clone(),
+ let mut validated = preprocess_and_validate(dot_source, goal_override, &TransformOptions {
current_dir,
file_resolver,
- Vec::new(),
- template_context,
- goal_override,
- RenderMode::Structural,
- options
- .settings
- .run
- .model
- .provider
- .as_deref()
- .filter(|provider| !provider.is_empty())
- .map(ProviderId::new),
- &options.configured_providers,
- false,
- &options.catalog,
- )?;
+ template_context: template_context(Some(&options.settings), vars),
+ source_name: options.source_name.clone(),
+ render_mode: RenderMode::Structural,
+ custom_transforms: Vec::new(),
+ model_resolution: Some(
+ ModelResolutionTransform::for_eligible(
+ Arc::clone(&options.catalog),
+ options.configured_providers.iter().cloned().collect(),
+ )
+ .with_default_provider(configured_default_provider(&options.settings)),
+ ),
+ })?;
validated.promote_template_undefined_variables_to_errors();
if validated.has_errors() {
@@ -326,36 +319,37 @@ fn create_from_source(
persist_validated(validated, options)
}
+/// Parse, transform, and validate `dot_source`.
+///
+/// `options.model_resolution` drives both halves of catalog awareness: it
+/// selects concrete models during TRANSFORM and enables the catalog-backed
+/// lint rules during VALIDATE. `None` leaves authored model and provider
+/// selectors untouched for offline structural validation.
pub(super) fn preprocess_and_validate(
dot_source: &str,
- source_name: Option,
- current_dir: Option,
- file_resolver: Option>,
- custom_transforms: Vec>,
- template_context: TemplateContext,
goal_override: Option<&str>,
- render_mode: RenderMode,
- default_provider: Option,
- eligible_providers: &[ProviderId],
- catalog_fallback: bool,
- catalog: &Arc,
+ options: &TransformOptions,
) -> Result {
let mut parsed = pipeline::parse(dot_source)?;
apply_goal_override(&mut parsed.graph, goal_override);
- let transformed = pipeline::transform(parsed, &TransformOptions {
- current_dir,
- file_resolver,
- template_context,
- source_name,
- render_mode,
- custom_transforms,
- catalog: Arc::clone(catalog),
- default_provider,
- eligible_providers: eligible_providers.iter().cloned().collect(),
- catalog_fallback,
- })?;
- Ok(pipeline::validate(transformed, catalog.as_ref(), &[]))
+ let transformed = pipeline::transform(parsed, options)?;
+ let catalog = options
+ .model_resolution
+ .as_ref()
+ .map(ModelResolutionTransform::catalog);
+ Ok(pipeline::validate(transformed, catalog, &[]))
+}
+
+/// The workflow-level default provider, treating an empty setting as unset.
+pub(super) fn configured_default_provider(settings: &WorkflowSettings) -> Option {
+ settings
+ .run
+ .model
+ .provider
+ .as_deref()
+ .filter(|provider| !provider.is_empty())
+ .map(ProviderId::new)
}
pub(super) fn template_context(
@@ -462,8 +456,9 @@ mod tests {
use object_store::memory::InMemory;
use super::*;
- use crate::operations::{ValidateInput, validate};
+ use crate::operations::{ValidateInput, validate, validate_with_catalog};
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
+ use crate::transforms::Transform;
use crate::workflow_bundle::BundledWorkflow;
fn memory_store() -> Arc {
Arc::new(Database::new(
@@ -553,17 +548,19 @@ reasoning = false
}
fn validate_dot(dot_source: &str, settings: WorkflowSettings) -> Validated {
- validate(ValidateInput {
- workflow: WorkflowInput::DotSource {
- source: dot_source.to_string(),
- base_dir: None,
+ validate_with_catalog(
+ ValidateInput {
+ workflow: WorkflowInput::DotSource {
+ source: dot_source.to_string(),
+ base_dir: None,
+ },
+ settings,
+ vars: HashMap::new(),
+ cwd: PathBuf::from("."),
+ custom_transforms: Vec::new(),
},
- settings,
- vars: HashMap::new(),
- cwd: PathBuf::from("."),
- custom_transforms: Vec::new(),
- catalog: test_catalog(),
- })
+ test_catalog(),
+ )
.unwrap()
}
@@ -573,21 +570,35 @@ reasoning = false
fn validate_dot_with_vars(dot_source: &str, vars: HashMap) -> Validated {
preprocess_and_validate(
dot_source,
- Some("workflow.fabro".to_string()),
- Some(PathBuf::from(".")),
None,
- Vec::new(),
- template_context(Some(&WorkflowSettings::default()), vars),
- None,
- RenderMode::Structural,
- None,
- &test_provider_ids(),
- false,
- &test_catalog(),
+ &test_transform_options(
+ PathBuf::from("."),
+ None,
+ RenderMode::Structural,
+ template_context(Some(&WorkflowSettings::default()), vars),
+ ),
)
.unwrap()
}
+ /// Catalog-backed TRANSFORM options for the built-in test catalog.
+ fn test_transform_options(
+ current_dir: PathBuf,
+ file_resolver: Option>,
+ render_mode: RenderMode,
+ template_context: TemplateContext,
+ ) -> TransformOptions {
+ TransformOptions {
+ current_dir: Some(current_dir),
+ file_resolver,
+ template_context,
+ source_name: Some("workflow.fabro".to_string()),
+ render_mode,
+ custom_transforms: Vec::new(),
+ model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
+ }
+ }
+
const MINIMAL_DOT: &str = r#"digraph Test {
graph [goal="Build feature"]
start [shape=Mdiamond]
@@ -744,17 +755,13 @@ reasoning = false
let result = preprocess_and_validate(
dot,
- Some("workflow.fabro".to_string()),
- Some(PathBuf::from(".")),
None,
- Vec::new(),
- template_context(Some(&WorkflowSettings::default()), HashMap::new()),
- None,
- RenderMode::Strict,
- None,
- &test_provider_ids(),
- false,
- &test_catalog(),
+ &test_transform_options(
+ PathBuf::from("."),
+ None,
+ RenderMode::Strict,
+ template_context(Some(&WorkflowSettings::default()), HashMap::new()),
+ ),
);
let Err(err) = result else {
panic!("expected strict mode to hard-fail on unbound inline prompt");
@@ -781,19 +788,15 @@ reasoning = false
let result = preprocess_and_validate(
dot,
- Some("workflow.fabro".to_string()),
- Some(dir.path().to_path_buf()),
- Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
- None,
- ))),
- Vec::new(),
- template_context(Some(&WorkflowSettings::default()), HashMap::new()),
None,
- RenderMode::Strict,
- None,
- &test_provider_ids(),
- false,
- &test_catalog(),
+ &test_transform_options(
+ dir.path().to_path_buf(),
+ Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
+ None,
+ ))),
+ RenderMode::Strict,
+ template_context(Some(&WorkflowSettings::default()), HashMap::new()),
+ ),
);
let Err(err) = result else {
panic!("expected strict mode to hard-fail on unbound imported prompt");
@@ -852,7 +855,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
});
assert!(result.is_err());
@@ -901,7 +903,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
let file_missing = validate(ValidateInput {
@@ -920,7 +921,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
assert_eq!(
@@ -944,7 +944,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
let file_goal = validate(ValidateInput {
@@ -963,7 +962,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
assert_eq!(
@@ -1055,7 +1053,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
});
assert!(result.is_err());
}
@@ -1100,7 +1097,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: vec![Box::new(TagTransform)],
- catalog: test_catalog(),
})
.unwrap();
validated.raise_on_errors().unwrap();
@@ -1135,7 +1131,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
validated.raise_on_errors().unwrap();
@@ -1177,7 +1172,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
@@ -1227,7 +1221,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
@@ -1278,7 +1271,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
- catalog: test_catalog(),
})
.unwrap();
diff --git a/lib/components/fabro-workflow/src/operations/mod.rs b/lib/components/fabro-workflow/src/operations/mod.rs
index 38b7d9aaa..9523c5ff5 100644
--- a/lib/components/fabro-workflow/src/operations/mod.rs
+++ b/lib/components/fabro-workflow/src/operations/mod.rs
@@ -22,7 +22,7 @@ pub use rewind::{RewindInput, RewindOutcome, rewind};
pub use source::WorkflowInput;
pub use start::{StartServices, Started, start};
pub use timeline::{ForkTarget, RunTimeline, TimelineEntry, build_timeline, timeline};
-pub use validate::{ValidateInput, validate, validate_with_ready_providers};
+pub use validate::{ValidateInput, validate, validate_with_catalog, validate_with_ready_providers};
pub use crate::pipeline::{LlmSpec, SandboxEnvSpec};
pub use crate::transforms::RenderMode;
diff --git a/lib/components/fabro-workflow/src/operations/validate.rs b/lib/components/fabro-workflow/src/operations/validate.rs
index c2b990f5c..dfcb38777 100644
--- a/lib/components/fabro-workflow/src/operations/validate.rs
+++ b/lib/components/fabro-workflow/src/operations/validate.rs
@@ -5,12 +5,12 @@ use std::sync::Arc;
use fabro_model::{Catalog, ProviderId};
use fabro_types::WorkflowSettings;
-use super::create::{preprocess_and_validate, template_context};
+use super::create::{configured_default_provider, preprocess_and_validate, template_context};
use super::source::{ResolveWorkflowInput, WorkflowInput, resolve_workflow};
use crate::error::Error;
use crate::operations::RenderMode;
-use crate::pipeline::Validated;
-use crate::transforms::Transform;
+use crate::pipeline::{TransformOptions, Validated};
+use crate::transforms::{ModelResolutionTransform, Transform};
pub struct ValidateInput {
pub workflow: WorkflowInput,
@@ -20,20 +20,24 @@ pub struct ValidateInput {
pub vars: HashMap,
pub cwd: PathBuf,
pub custom_transforms: Vec>,
- pub catalog: Arc,
}
-/// Parse, transform, and validate a DOT source string.
+/// Parse, transform, and structurally validate a DOT source string without a
+/// model catalog. Model and provider availability is left to the caller that
+/// owns a catalog — typically the server.
///
/// Returns `Validated` even when validation produced errors. Call
/// `validated.raise_on_errors()` if the caller wants to fail fast.
pub fn validate(input: ValidateInput) -> Result {
- let eligible_providers = input
- .catalog
- .all_provider_ids()
- .into_iter()
- .collect::>();
- validate_with_eligible_providers(input, &eligible_providers, false)
+ validate_resolving_models(input, None)
+}
+
+/// Parse, transform, and validate a DOT source string against `catalog`.
+pub fn validate_with_catalog(
+ input: ValidateInput,
+ catalog: Arc,
+) -> Result {
+ validate_resolving_models(input, Some(ModelResolutionTransform::new(catalog)))
}
/// Parse, transform, and validate, resolving models against the ready
@@ -41,45 +45,60 @@ pub fn validate(input: ValidateInput) -> Result {
/// provider-readiness selection failures.
pub fn validate_with_ready_providers(
input: ValidateInput,
+ catalog: Arc,
ready_providers: &[ProviderId],
) -> Result {
- validate_with_eligible_providers(input, ready_providers, true)
+ validate_resolving_models(
+ input,
+ Some(
+ ModelResolutionTransform::for_eligible(
+ catalog,
+ ready_providers.iter().cloned().collect(),
+ )
+ .with_catalog_fallback(true),
+ ),
+ )
}
-fn validate_with_eligible_providers(
+/// The workflow's own default provider is only known once the workflow is
+/// resolved, so callers hand in a partially built transform and it is
+/// completed here.
+fn validate_resolving_models(
input: ValidateInput,
- eligible_providers: &[ProviderId],
- catalog_fallback: bool,
+ model_resolution: Option,
) -> Result {
+ let ValidateInput {
+ workflow,
+ settings,
+ vars,
+ cwd,
+ custom_transforms,
+ } = input;
let resolved = resolve_workflow(ResolveWorkflowInput {
- workflow: input.workflow,
- settings: input.settings,
- cwd: input.cwd,
+ workflow,
+ settings,
+ cwd,
})
.map_err(|err| Error::Parse(err.to_string()))?;
+ let model_resolution = model_resolution.map(|resolution| {
+ resolution.with_default_provider(configured_default_provider(&resolved.settings))
+ });
+
preprocess_and_validate(
&resolved.raw_source,
- resolved
- .dot_path
- .as_ref()
- .map(|path| path.display().to_string()),
- resolved.current_dir,
- resolved.file_resolver,
- input.custom_transforms,
- template_context(Some(&resolved.settings), input.vars),
resolved.goal_override.as_deref(),
- RenderMode::Structural,
- resolved
- .settings
- .run
- .model
- .provider
- .as_deref()
- .filter(|provider| !provider.is_empty())
- .map(fabro_model::ProviderId::new),
- eligible_providers,
- catalog_fallback,
- &input.catalog,
+ &TransformOptions {
+ current_dir: resolved.current_dir,
+ file_resolver: resolved.file_resolver,
+ template_context: template_context(Some(&resolved.settings), vars),
+ source_name: resolved
+ .dot_path
+ .as_ref()
+ .map(|path| path.display().to_string()),
+ render_mode: RenderMode::Structural,
+ custom_transforms,
+ model_resolution,
+ },
)
}
diff --git a/lib/components/fabro-workflow/src/pipeline/transform.rs b/lib/components/fabro-workflow/src/pipeline/transform.rs
index b399d6637..bf3b82471 100644
--- a/lib/components/fabro-workflow/src/pipeline/transform.rs
+++ b/lib/components/fabro-workflow/src/pipeline/transform.rs
@@ -3,8 +3,8 @@ use std::sync::Arc;
use super::types::{Parsed, TransformOptions, Transformed};
use crate::error::Error;
use crate::transforms::{
- FileInliningTransform, ImportTransform, ModelResolutionTransform,
- StylesheetApplicationTransform, TemplateTransform, Transform,
+ FileInliningTransform, ImportTransform, StylesheetApplicationTransform, TemplateTransform,
+ Transform,
};
/// TRANSFORM phase: apply built-in and custom transforms to a parsed graph.
@@ -63,13 +63,10 @@ pub fn transform(parsed: Parsed, options: &TransformOptions) -> Result model_resolution.apply(graph)?,
+ None => graph,
+ };
// Custom transforms
let graph = options
@@ -98,6 +95,7 @@ mod tests {
use crate::file_resolver::FilesystemFileResolver;
use crate::pipeline::parse::parse;
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
+ use crate::transforms::ModelResolutionTransform;
fn write_file(path: &Path, contents: &str) {
if let Some(parent) = path.parent() {
@@ -112,16 +110,13 @@ mod tests {
fn transform_options() -> TransformOptions {
TransformOptions {
- current_dir: None,
- file_resolver: None,
- template_context: fabro_template::TemplateContext::new(),
- source_name: None,
- render_mode: crate::operations::RenderMode::Strict,
- custom_transforms: vec![],
- catalog: test_catalog(),
- default_provider: None,
- eligible_providers: Catalog::builtin().all_provider_ids(),
- catalog_fallback: false,
+ current_dir: None,
+ file_resolver: None,
+ template_context: fabro_template::TemplateContext::new(),
+ source_name: None,
+ render_mode: crate::operations::RenderMode::Strict,
+ custom_transforms: vec![],
+ model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
}
}
@@ -177,16 +172,9 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
- current_dir: Some(dir.path().to_path_buf()),
- file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
- template_context: fabro_template::TemplateContext::new(),
- source_name: None,
- render_mode: crate::operations::RenderMode::Strict,
- custom_transforms: vec![],
- catalog: test_catalog(),
- default_provider: None,
- eligible_providers: Catalog::builtin().all_provider_ids(),
- catalog_fallback: false,
+ current_dir: Some(dir.path().to_path_buf()),
+ file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
+ ..transform_options()
})
.unwrap();
@@ -227,21 +215,15 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
- current_dir: Some(dir.path().to_path_buf()),
- file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
- template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from(
- [(
+ current_dir: Some(dir.path().to_path_buf()),
+ file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
+ template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from([
+ (
"task".to_string(),
toml::Value::String("Launch".to_string()),
- )],
- )),
- source_name: None,
- render_mode: crate::operations::RenderMode::Strict,
- custom_transforms: vec![],
- catalog: test_catalog(),
- default_provider: None,
- eligible_providers: Catalog::builtin().all_provider_ids(),
- catalog_fallback: false,
+ ),
+ ])),
+ ..transform_options()
})
.unwrap();
@@ -349,6 +331,33 @@ mod tests {
);
}
+ #[test]
+ fn structural_transform_preserves_catalog_owned_model_selection() {
+ let dot = r#"digraph Test {
+ graph [goal="Test"]
+ start [shape=Mdiamond]
+ work [prompt="Do work", model="private-model", provider="server-only"]
+ exit [shape=Msquare]
+ start -> work -> exit
+ }"#;
+ let parsed = parse(dot).unwrap();
+ let transformed = transform(parsed, &TransformOptions {
+ model_resolution: None,
+ ..transform_options()
+ })
+ .unwrap();
+ let work = &transformed.graph.nodes["work"];
+
+ assert_eq!(
+ work.attrs.get("model").and_then(AttrValue::as_str),
+ Some("private-model")
+ );
+ assert_eq!(
+ work.attrs.get("provider").and_then(AttrValue::as_str),
+ Some("server-only")
+ );
+ }
+
#[test]
fn transform_reports_goal_self_reference_once_across_passes() {
// FileInlining renders the goal for prompt context, but TemplateTransform
@@ -365,16 +374,10 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
- current_dir: Some(dir.path().to_path_buf()),
- file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
- template_context: fabro_template::TemplateContext::new(),
- source_name: None,
- render_mode: crate::operations::RenderMode::Structural,
- custom_transforms: vec![],
- catalog: test_catalog(),
- default_provider: None,
- eligible_providers: Catalog::builtin().all_provider_ids(),
- catalog_fallback: false,
+ current_dir: Some(dir.path().to_path_buf()),
+ file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
+ render_mode: crate::operations::RenderMode::Structural,
+ ..transform_options()
})
.unwrap();
diff --git a/lib/components/fabro-workflow/src/pipeline/types.rs b/lib/components/fabro-workflow/src/pipeline/types.rs
index 425cba9aa..9ff9b83fd 100644
--- a/lib/components/fabro-workflow/src/pipeline/types.rs
+++ b/lib/components/fabro-workflow/src/pipeline/types.rs
@@ -1,4 +1,4 @@
-use std::collections::{HashMap, HashSet};
+use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::Arc;
@@ -28,7 +28,7 @@ use crate::runtime_store::RunStoreHandle;
use crate::services::{EngineServices, FabroRunToolServices, RunServices};
use crate::stage_execution::StageExecutionSeed;
use crate::steering_hub::SteeringHub;
-use crate::transforms::{RenderMode, Transform};
+use crate::transforms::{ModelResolutionTransform, RenderMode, Transform};
use crate::workflow_bundle::WorkflowBundle;
/// Output of the PARSE phase.
@@ -359,18 +359,15 @@ pub struct Finalized {
/// Options for the TRANSFORM phase.
pub struct TransformOptions {
- pub current_dir: Option,
- pub file_resolver: Option>,
- pub template_context: TemplateContext,
- pub source_name: Option,
- pub render_mode: RenderMode,
- pub custom_transforms: Vec>,
- pub catalog: Arc,
- pub default_provider: Option,
- pub eligible_providers: HashSet,
- /// Fall back to the full catalog when the eligible providers cannot
- /// supply a requested model, instead of erroring.
- pub catalog_fallback: bool,
+ pub current_dir: Option,
+ pub file_resolver: Option>,
+ pub template_context: TemplateContext,
+ pub source_name: Option,
+ pub render_mode: RenderMode,
+ pub custom_transforms: Vec>,
+ /// Catalog-backed model resolution to perform. `None` preserves authored
+ /// model and provider selectors for catalog-free structural validation.
+ pub model_resolution: Option,
}
/// Options for the FINALIZE phase.
diff --git a/lib/components/fabro-workflow/src/pipeline/validate.rs b/lib/components/fabro-workflow/src/pipeline/validate.rs
index f0cd51dbd..32bfaf69c 100644
--- a/lib/components/fabro-workflow/src/pipeline/validate.rs
+++ b/lib/components/fabro-workflow/src/pipeline/validate.rs
@@ -5,11 +5,15 @@ use super::types::{Transformed, Validated};
/// VALIDATE phase: run lint rules against the transformed graph.
///
+/// Catalog-backed rules (model and provider availability) run only when
+/// `catalog` is `Some`. Offline callers pass `None` so a workflow naming a
+/// server-owned model is left for the server to judge.
+///
/// **Infallible.** Always returns `Validated` with diagnostics. Caller decides
/// whether to fail via `validated.raise_on_errors()`.
pub fn validate(
transformed: Transformed,
- catalog: &Catalog,
+ catalog: Option<&Catalog>,
extra_rules: &[&dyn LintRule],
) -> Validated {
let Transformed {
@@ -17,44 +21,33 @@ pub fn validate(
source,
mut diagnostics,
} = transformed;
- diagnostics.extend(fabro_validate::validate_with_catalog(
- &graph,
- catalog,
- extra_rules,
- ));
+ diagnostics.extend(match catalog {
+ Some(catalog) => fabro_validate::validate_with_catalog(&graph, catalog, extra_rules),
+ None => fabro_validate::validate(&graph, extra_rules),
+ });
Validated::new(graph, source, diagnostics)
}
#[cfg(test)]
mod tests {
- use fabro_model::Catalog;
-
use super::*;
use crate::pipeline::parse::parse;
use crate::pipeline::transform;
use crate::pipeline::types::TransformOptions;
- fn test_catalog() -> std::sync::Arc {
- std::sync::Arc::new(Catalog::from_builtin().unwrap())
- }
-
fn run_pipeline(dot: &str) -> Validated {
- let catalog = test_catalog();
let parsed = parse(dot).unwrap();
let transformed = transform::transform(parsed, &TransformOptions {
- current_dir: None,
- file_resolver: None,
- template_context: fabro_template::TemplateContext::new(),
- source_name: None,
- render_mode: crate::operations::RenderMode::Strict,
- custom_transforms: vec![],
- catalog: std::sync::Arc::clone(&catalog),
- default_provider: None,
- eligible_providers: catalog.all_provider_ids(),
- catalog_fallback: false,
+ current_dir: None,
+ file_resolver: None,
+ template_context: fabro_template::TemplateContext::new(),
+ source_name: None,
+ render_mode: crate::operations::RenderMode::Strict,
+ custom_transforms: vec![],
+ model_resolution: None,
})
.unwrap();
- validate(transformed, catalog.as_ref(), &[])
+ validate(transformed, None, &[])
}
#[test]
diff --git a/lib/components/fabro-workflow/src/transforms/import.rs b/lib/components/fabro-workflow/src/transforms/import.rs
index b5406ddc6..f2c5e74fd 100644
--- a/lib/components/fabro-workflow/src/transforms/import.rs
+++ b/lib/components/fabro-workflow/src/transforms/import.rs
@@ -339,18 +339,18 @@ impl ImportTransform {
.edges
.retain(|edge| edge.from != placeholder_id && edge.to != placeholder_id);
- for (node_id, node) in imported_graph.nodes {
+ for (node_id, mut merged_node) in imported_graph.nodes {
if node_id == start_id || node_id == exit_id {
continue;
}
let prefixed_id = format!("{placeholder_id}.{node_id}");
- let mut merged_node = Node::new(&prefixed_id);
+ merged_node.id.clone_from(&prefixed_id);
+ let imported_attrs = std::mem::take(&mut merged_node.attrs);
merged_node.attrs.clone_from(&placeholder.default_attrs);
- merged_node.attrs.extend(node.attrs);
+ merged_node.attrs.extend(imported_attrs);
Self::remap_retry_target(&mut merged_node.attrs, placeholder_id);
- merged_node.classes = node.classes;
for class_name in &placeholder.class_names {
Self::push_class(&mut merged_node.classes, class_name);
}
@@ -649,7 +649,10 @@ impl PreparedImport {
self.graph.nodes.iter().all(|(node_id, node)| {
ImportTransform::is_start_sentinel(node_id, node)
|| ImportTransform::is_exit_sentinel(node_id, node)
- })
+ }) && matches!(
+ self.graph.edges.as_slice(),
+ [edge] if edge.from == self.start_id && edge.to == self.exit_id
+ )
}
}
@@ -909,6 +912,7 @@ mod tests {
assert!(!graph.nodes.contains_key("validate"));
assert!(graph.nodes.contains_key("validate.lint"));
assert!(graph.nodes.contains_key("validate.test"));
+ assert_eq!(graph.nodes["validate.lint"].id, "validate.lint");
assert!(!graph.nodes.contains_key("validate.start"));
assert!(!graph.nodes.contains_key("validate.exit"));
@@ -951,6 +955,72 @@ mod tests {
);
}
+ #[test]
+ fn edge_only_node_in_imported_fragment_stays_missing() {
+ let dir = tempfile::tempdir().unwrap();
+ write_file(
+ &dir.path().join("validate.fabro"),
+ r#"digraph validate {
+ start [shape=Mdiamond]
+ lint [prompt="Run clippy"]
+ exit [shape=Msquare]
+ start -> lint -> typo -> exit
+ }"#,
+ );
+
+ let graph = apply_import(
+ r#"digraph Deploy {
+ start [shape=Mdiamond]
+ validate [import="./validate.fabro"]
+ exit [shape=Msquare]
+ start -> validate -> exit
+ }"#,
+ dir.path(),
+ None,
+ );
+
+ assert!(!graph.nodes.contains_key("validate.typo"));
+ assert!(graph.nodes.contains_key("validate.lint"));
+ }
+
+ #[test]
+ fn edge_only_body_is_not_treated_as_empty_import() {
+ let dir = tempfile::tempdir().unwrap();
+ write_file(
+ &dir.path().join("validate.fabro"),
+ r"digraph validate {
+ start [shape=Mdiamond]
+ exit [shape=Msquare]
+ start -> typo -> exit
+ }",
+ );
+
+ let graph = apply_import(
+ r#"digraph Deploy {
+ start [shape=Mdiamond]
+ validate [import="./validate.fabro"]
+ exit [shape=Msquare]
+ start -> validate -> exit
+ }"#,
+ dir.path(),
+ None,
+ );
+
+ assert!(!graph.nodes.contains_key("validate.typo"));
+ assert!(
+ graph
+ .edges
+ .iter()
+ .any(|edge| edge.from == "start" && edge.to == "validate.typo")
+ );
+ assert!(
+ graph
+ .edges
+ .iter()
+ .any(|edge| edge.from == "validate.typo" && edge.to == "exit")
+ );
+ }
+
#[test]
fn import_reports_structural_diagnostic_for_imported_prompt_templates() {
let dir = tempfile::tempdir().unwrap();
@@ -1130,6 +1200,48 @@ mod tests {
);
}
+ #[test]
+ fn imported_start_and_exit_sentinels_must_be_declared() {
+ let dir = tempfile::tempdir().unwrap();
+ let host = r#"digraph Deploy {
+ start [shape=Mdiamond]
+ validate [import="./validate.fabro"]
+ exit [shape=Msquare]
+ start -> validate -> exit
+ }"#;
+ let cases = [
+ (
+ r#"digraph validate {
+ work [prompt="Run checks"]
+ exit [shape=Msquare]
+ start -> work -> exit
+ }"#,
+ "imported workflow must have exactly one start node, found 0",
+ ),
+ (
+ r#"digraph validate {
+ start [shape=Mdiamond]
+ work [prompt="Run checks"]
+ start -> work -> exit
+ }"#,
+ "imported workflow must have exactly one exit node, found 0",
+ ),
+ ];
+
+ for (source, expected_error) in cases {
+ write_file(&dir.path().join("validate.fabro"), source);
+ let graph = apply_import(host, dir.path(), None);
+
+ assert_eq!(
+ graph.nodes["validate"]
+ .attrs
+ .get("import_error")
+ .and_then(AttrValue::as_str),
+ Some(expected_error)
+ );
+ }
+ }
+
#[test]
fn multiple_entry_nodes_poison_placeholder() {
let dir = tempfile::tempdir().unwrap();
diff --git a/lib/components/fabro-workflow/src/transforms/model_resolution.rs b/lib/components/fabro-workflow/src/transforms/model_resolution.rs
index 12f29ac5a..7fb61b9d9 100644
--- a/lib/components/fabro-workflow/src/transforms/model_resolution.rs
+++ b/lib/components/fabro-workflow/src/transforms/model_resolution.rs
@@ -52,6 +52,13 @@ impl ModelResolutionTransform {
self
}
+ /// The catalog this transform resolves against, so callers can run the
+ /// matching catalog-backed lint rules.
+ #[must_use]
+ pub fn catalog(&self) -> &Catalog {
+ &self.catalog
+ }
+
fn resolve_model(
&self,
model: &str,
diff --git a/lib/components/fabro-workflow/src/transforms/variable_expansion.rs b/lib/components/fabro-workflow/src/transforms/variable_expansion.rs
index 66434266a..e739c21a4 100644
--- a/lib/components/fabro-workflow/src/transforms/variable_expansion.rs
+++ b/lib/components/fabro-workflow/src/transforms/variable_expansion.rs
@@ -19,17 +19,18 @@ use crate::static_reference::{
/// How the template-expansion pass should treat undefined input variables.
///
-/// Validate is structural — it should not fail just because the user has not
-/// bound `{{ inputs.* }}` yet. Run-start is strict — missing inputs are real
-/// errors. Splitting the two lets validate work on a bare `.fabro` while
-/// run-start preserves its current hard-fail behavior.
+/// Both validate and run-create render structurally so they can report every
+/// unbound `{{ inputs.* }}` variable in one pass rather than aborting on the
+/// first. Run-create then promotes the resulting warnings to errors, which
+/// keeps its hard-fail behavior.
#[derive(Clone, Copy, Debug)]
pub enum RenderMode {
- /// Undefined inputs are hard errors. Used by run-create.
+ /// Undefined inputs abort the pass with a hard error. No production caller
+ /// uses this today; run-create promotes structural warnings instead.
Strict,
/// Undefined inputs render as empty and become warning diagnostics on the
/// returned `Validated`, so structural lints still run. Used by
- /// `fabro validate`.
+ /// `fabro validate` and by run-create.
Structural,
}
diff --git a/lib/components/fabro-workflow/tests/it/integration.rs b/lib/components/fabro-workflow/tests/it/integration.rs
index dc13bcbf8..e4f64f17c 100644
--- a/lib/components/fabro-workflow/tests/it/integration.rs
+++ b/lib/components/fabro-workflow/tests/it/integration.rs
@@ -388,6 +388,9 @@ fn parse_and_validate_human_gate() {
type="human"
]
+ ship_it [prompt="Ship the change"]
+ fixes [prompt="Apply the requested fixes"]
+
start -> review_gate
review_gate -> ship_it [label="[A] Approve"]
review_gate -> fixes [label="[F] Fix"]
@@ -4854,6 +4857,7 @@ async fn manager_loop_child_workflow_e2e() {
#[tokio::test]
async fn import_e2e_through_engine() {
use fabro_workflow::pipeline::{TransformOptions, transform, validate};
+ use fabro_workflow::transforms::ModelResolutionTransform;
let dir = tempfile::tempdir().unwrap();
let catalog = std::sync::Arc::new(
@@ -4897,21 +4901,20 @@ async fn import_e2e_through_engine() {
)
.expect("parse should succeed");
let transformed = transform(parsed, &TransformOptions {
- current_dir: Some(dir.path().to_path_buf()),
- file_resolver: Some(std::sync::Arc::new(
+ current_dir: Some(dir.path().to_path_buf()),
+ file_resolver: Some(std::sync::Arc::new(
fabro_workflow::file_resolver::FilesystemFileResolver::new(None),
)),
- template_context: fabro_template::TemplateContext::new(),
- source_name: None,
- render_mode: fabro_workflow::operations::RenderMode::Strict,
- custom_transforms: vec![],
- catalog: std::sync::Arc::clone(&catalog),
- default_provider: None,
- eligible_providers: catalog.all_provider_ids(),
- catalog_fallback: false,
+ template_context: fabro_template::TemplateContext::new(),
+ source_name: None,
+ render_mode: fabro_workflow::operations::RenderMode::Strict,
+ custom_transforms: vec![],
+ model_resolution: Some(ModelResolutionTransform::new(std::sync::Arc::clone(
+ &catalog,
+ ))),
})
.unwrap();
- let validated = validate(transformed, catalog.as_ref(), &[]);
+ let validated = validate(transformed, Some(catalog.as_ref()), &[]);
validated
.raise_on_errors()
.expect("validation should pass after imports expand");
diff --git a/lib/foundation/fabro-api/build.rs b/lib/foundation/fabro-api/build.rs
index fc943e99d..740585545 100644
--- a/lib/foundation/fabro-api/build.rs
+++ b/lib/foundation/fabro-api/build.rs
@@ -361,6 +361,11 @@ fn main() {
"fabro_types::StageInferenceProjection",
&[],
),
+ (
+ "StageToolBatchProjection",
+ "fabro_types::StageToolBatchProjection",
+ &[],
+ ),
("LlmOutputKind", "fabro_types::LlmOutputKind", &[]),
("PermissionLevel", "fabro_types::PermissionLevel", &[]),
(
diff --git a/lib/foundation/fabro-api/src/lib.rs b/lib/foundation/fabro-api/src/lib.rs
index 4312acb91..b9a6a7185 100644
--- a/lib/foundation/fabro-api/src/lib.rs
+++ b/lib/foundation/fabro-api/src/lib.rs
@@ -69,10 +69,10 @@ pub mod types {
StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
StageContextWindowStaleness, StageContextWindowUnavailableReason,
StageContextWindowWarning, StageHandler, StageId, StageInferenceProjection,
- StageModelUsage, StageOutcome, StageProjection, StageState, SubAgentProjection,
- SubAgentStatus, SystemActorKind, SystemIntegrationStatus, SystemIntegrationsResponse,
- TodoListProjection, TurnId, UpdateVariableRequest, UserPrincipal, Variable,
- VariableListResponse, WorkflowSettings,
+ StageModelUsage, StageOutcome, StageProjection, StageState, StageToolBatchProjection,
+ SubAgentProjection, SubAgentStatus, SystemActorKind, SystemIntegrationStatus,
+ SystemIntegrationsResponse, TodoListProjection, TurnId, UpdateVariableRequest,
+ UserPrincipal, Variable, VariableListResponse, WorkflowSettings,
};
pub use crate::generated::types::*;
diff --git a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs
index 5faed1bd3..602b6c8c7 100644
--- a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs
+++ b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs
@@ -18,6 +18,7 @@ use fabro_api::types::{
StageContextWindowUnavailableReason as ApiStageContextWindowUnavailableReason,
StageContextWindowWarning as ApiStageContextWindowWarning,
StageInferenceProjection as ApiStageInferenceProjection, StageProjection as ApiStageProjection,
+ StageToolBatchProjection as ApiStageToolBatchProjection,
SubAgentProjection as ApiSubAgentProjection, SubAgentStatus as ApiSubAgentStatus,
TodoListProjection as ApiTodoListProjection,
};
@@ -29,8 +30,8 @@ use fabro_types::{
ParallelBranchResult, PermissionLevel, SkillsProjection, StageContextWindow,
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
- StageContextWindowWarning, StageInferenceProjection, StageProjection, SubAgentProjection,
- SubAgentStatus, TodoListKind, TodoListProjection,
+ StageContextWindowWarning, StageInferenceProjection, StageProjection, StageToolBatchProjection,
+ SubAgentProjection, SubAgentStatus, TodoListKind, TodoListProjection,
};
use serde_json::json;
@@ -68,9 +69,32 @@ fn stage_projection_reuses_nested_agent_state_types() {
);
assert_same_type::();
assert_same_type::();
+ assert_same_type::();
assert_same_type::();
}
+#[test]
+fn stage_tool_batch_projection_matches_openapi_json_shape() {
+ let batch = StageToolBatchProjection {
+ session_id: "ses_root".to_string(),
+ started_at: "2026-04-29T12:34:00Z".parse().unwrap(),
+ open_call_ids: ["call_1".to_string(), "call_2".to_string()]
+ .into_iter()
+ .collect(),
+ };
+ let value = serde_json::to_value(&batch).unwrap();
+ assert_eq!(
+ value,
+ json!({
+ "session_id": "ses_root",
+ "started_at": "2026-04-29T12:34:00Z",
+ "open_call_ids": ["call_1", "call_2"]
+ })
+ );
+ let api_batch: ApiStageToolBatchProjection = serde_json::from_value(value).unwrap();
+ assert_eq!(api_batch, batch);
+}
+
#[test]
fn stage_inference_projection_matches_openapi_json_shape() {
let inference = StageInferenceProjection {
@@ -148,6 +172,10 @@ fn stage_projection_without_inference_round_trips() {
let stage: StageProjection = serde_json::from_value(value.clone()).unwrap();
assert!(stage.inference.is_none());
+ assert!(stage.acp_started_at.is_none());
+ assert!(stage.tool_batch.is_none());
+ assert_eq!(stage.live_inference_ms, 0);
+ assert_eq!(stage.live_tool_ms, 0);
assert_eq!(serde_json::to_value(stage).unwrap(), value);
}
@@ -311,6 +339,7 @@ fn stage_projection_round_trips_representative_json() {
"first_output_kind": "text",
"retries": 0
},
+ "acp_started_at": "2026-04-29T12:34:00Z",
"agent_control": "running",
"state": "succeeded"
});
diff --git a/lib/foundation/fabro-types/src/lib.rs b/lib/foundation/fabro-types/src/lib.rs
index 68f5644a2..b7f3b0474 100644
--- a/lib/foundation/fabro-types/src/lib.rs
+++ b/lib/foundation/fabro-types/src/lib.rs
@@ -125,7 +125,7 @@ pub use run_projection::{
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
StageContextWindowWarning, StageInferenceProjection, StageModelUsage, StageProjection,
- SubAgentProjection, SubAgentStatus, first_event_seq,
+ StageToolBatchProjection, SubAgentProjection, SubAgentStatus, first_event_seq,
};
pub use run_sandbox::{
RunSandbox, RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan,
diff --git a/lib/foundation/fabro-types/src/run_event/mod.rs b/lib/foundation/fabro-types/src/run_event/mod.rs
index d4c8e55c1..6680fff2e 100644
--- a/lib/foundation/fabro-types/src/run_event/mod.rs
+++ b/lib/foundation/fabro-types/src/run_event/mod.rs
@@ -926,8 +926,6 @@ impl<'de> Deserialize<'de> for RunEvent {
#[cfg(test)]
mod tests {
- use std::collections::HashMap;
-
use serde_json::json;
use super::*;
@@ -992,20 +990,9 @@ mod tests {
#[test]
fn run_event_deserializes_adjacent_layout() {
let settings = WorkflowSettings::default();
- let graph = Graph {
- name: "test".to_string(),
- nodes: HashMap::from([("start".to_string(), Node {
- id: "start".to_string(),
- attrs: HashMap::new(),
- classes: Vec::new(),
- })]),
- edges: vec![Edge {
- from: "start".to_string(),
- to: "done".to_string(),
- attrs: HashMap::new(),
- }],
- attrs: HashMap::new(),
- };
+ let mut graph = Graph::new("test");
+ graph.nodes.insert("start".to_string(), Node::new("start"));
+ graph.edges.push(Edge::new("start", "done"));
let line = json!({
"id": "evt_2",
diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs
index 027cce2fc..4fcf16778 100644
--- a/lib/foundation/fabro-types/src/run_projection.rs
+++ b/lib/foundation/fabro-types/src/run_projection.rs
@@ -1,9 +1,9 @@
use std::borrow::Cow;
-use std::collections::{BTreeMap, HashMap};
+use std::collections::{BTreeMap, BTreeSet, HashMap};
use std::num::NonZeroU32;
use chrono::{DateTime, Utc};
-use fabro_model::{ReasoningEffort, Speed};
+use fabro_model::{Catalog, ReasoningEffort, Speed};
use strum::{Display, EnumString, IntoStaticStr};
use crate::run_event::{AgentSessionActivatedProps, StagePromptProps};
@@ -12,7 +12,7 @@ use crate::{
AgentToolSummary, BilledTokenCounts, Checkpoint, Conclusion, InterviewQuestionRecord,
InvalidTransition, LlmOutputKind, ModelRef, PermissionLevel, PullRequestLink, RunApproval,
RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, RunTiming, StageCompletion,
- StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection,
+ StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection, timing,
};
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
@@ -354,10 +354,31 @@ pub struct StageProjection {
/// immutable projections under their own `StageId`s.
///
/// `None` for stages still in flight (`started_at` is set but no terminal
- /// event has been observed yet). For live wall-time ticking, the UI uses
- /// `started_at`; once terminal this carries the finalized breakdown.
+ /// event has been observed yet). For a live breakdown while in flight, use
+ /// [`StageProjection::live_timing`]; once terminal this carries the
+ /// finalized, authoritative breakdown.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub timing: Option,
+ /// Inference time accumulated from closed brackets during this attempt.
+ ///
+ /// Live estimate only: the authoritative value arrives with the terminal
+ /// event and lands in `timing`. Excludes the currently-open bracket, which
+ /// [`StageProjection::live_timing`] adds from `inference.started_at`.
+ #[serde(default, skip_serializing_if = "is_zero_ms")]
+ pub live_inference_ms: u64,
+ /// Tool time accumulated from closed tool batches during this attempt.
+ ///
+ /// A batch spans the first `agent.tool.started` with no outstanding calls
+ /// through the `agent.tool.completed` that drains the last one, so tools
+ /// running concurrently within a turn are counted once. This matches how
+ /// the in-process stopwatch brackets `execute_tool_calls`; summing
+ /// per-call durations would over-count parallel tool use.
+ #[serde(default, skip_serializing_if = "is_zero_ms")]
+ pub live_tool_ms: u64,
+ /// Open tool batch for this stage: when the current batch started, and the
+ /// `tool_call_id`s that have not yet reported completion.
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub tool_batch: Option,
#[serde(default)]
pub usage: BilledTokenCounts,
#[serde(default, skip_serializing_if = "Option::is_none")]
@@ -390,11 +411,45 @@ pub struct StageProjection {
/// the authority on whether a run is actually stuck.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub inference: Option,
+ /// Start of an external ACP agent process, if one is running.
+ ///
+ /// ACP agents do not emit Fabro's internal LLM brackets, so their process
+ /// lifetime is the best available live inference estimate.
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub acp_started_at: Option>,
#[serde(default)]
pub agent_control: AgentControlState,
pub state: StageState,
}
+/// Serde guard so zero-valued live accumulators stay off the wire.
+#[allow(
+ clippy::trivially_copy_pass_by_ref,
+ reason = "serde skip_serializing_if predicates receive fields by reference"
+)]
+fn is_zero_ms(value: &u64) -> bool {
+ *value == 0
+}
+
+/// One open tool batch: tool calls dispatched together that have not all
+/// reported completion.
+///
+/// `open_call_ids` is a set rather than a count because `agent.tool.completed`
+/// identifies its call by id, and a projection replaying a truncated or
+/// duplicated log must not let a repeated completion drain the batch early.
+#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
+pub struct StageToolBatchProjection {
+ /// Root agent session that dispatched the batch. Later transitions are
+ /// gated on it so delayed events from a replaced session cannot mutate
+ /// the current session's batch.
+ pub session_id: String,
+ /// When the batch opened — the first `agent.tool.started` observed while
+ /// no other calls were outstanding.
+ pub started_at: DateTime,
+ /// Calls dispatched but not yet completed, by `tool_call_id`.
+ pub open_call_ids: BTreeSet,
+}
+
/// One open inference bracket: a dispatched LLM request that has not yet
/// produced a message, error, or interrupt.
#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)]
@@ -506,6 +561,9 @@ impl StageProjection {
response: None,
completion: None,
timing: None,
+ live_inference_ms: 0,
+ live_tool_ms: 0,
+ tool_batch: None,
usage: BilledTokenCounts::default(),
model: None,
root_agent_todos: None,
@@ -516,6 +574,7 @@ impl StageProjection {
mcp_servers: Vec::new(),
context_window: None,
inference: None,
+ acp_started_at: None,
agent_control: AgentControlState::default(),
provider_used: None,
diff: None,
@@ -540,6 +599,29 @@ impl StageProjection {
self.state
}
+ /// This stage's token counts with a cost attached.
+ ///
+ /// A provider-reported cost always wins. Otherwise the catalog prices the
+ /// recorded tokens for the stage's model. The stored counts pass through
+ /// untouched when there is no catalog, no model, or no price for that
+ /// model. Empty usage also passes through untouched. These cases leave
+ /// `total_usd_micros` as `None` rather than zero.
+ #[must_use]
+ pub fn billed_usage(&self, catalog: Option<&Catalog>) -> Cow<'_, BilledTokenCounts> {
+ if self.usage.total_usd_micros.is_some() || self.usage.is_zero() {
+ return Cow::Borrowed(&self.usage);
+ }
+ let (Some(catalog), Some(model)) = (catalog, self.model.as_ref()) else {
+ return Cow::Borrowed(&self.usage);
+ };
+ let Some(total_usd_micros) = catalog.price_tokens(model, &self.usage.token_counts()) else {
+ return Cow::Borrowed(&self.usage);
+ };
+ let mut usage = self.usage.clone();
+ usage.total_usd_micros = Some(total_usd_micros);
+ Cow::Owned(usage)
+ }
+
/// Live wall-clock time in milliseconds.
///
/// While the stage is non-terminal (`Pending`, `Running`, or `Retrying`),
@@ -555,14 +637,197 @@ impl StageProjection {
state,
StageState::Running | StageState::Retrying | StageState::Pending
) {
- return self.started_at.map(|started| {
- u64::try_from(now.signed_duration_since(started).num_milliseconds().max(0))
- .unwrap_or(0)
- });
+ return self
+ .started_at
+ .map(|started| timing::elapsed_ms(started, now));
}
self.timing.map(|timing| timing.wall_time_ms)
}
+ /// Live timing breakdown in milliseconds — the active-time twin of
+ /// [`Self::live_wall_time_ms`].
+ ///
+ /// Once terminal, returns the stored `timing` unchanged: the finalized
+ /// breakdown comes from the worker's own stopwatch and is authoritative.
+ ///
+ /// While in flight, returns an estimate reconstructed from the event log:
+ /// accumulated closed brackets plus whatever bracket is open right now.
+ /// The estimate is per-handler, because only agent stages emit brackets at
+ /// all:
+ ///
+ /// - `Agent` — accumulated inference and tool brackets, plus the open
+ /// inference bracket and open tool batch.
+ /// - `Prompt` — one inference call spanning the stage, so elapsed time
+ /// since `started_at` counts as inference. Matches the finalized
+ /// `active_only(inference, 0)`.
+ /// - `Command` — the command *is* the work, so elapsed time counts as tool.
+ /// Matches the finalized `active_only(0, duration_ms)`.
+ /// - Everything else — zero. Waiting on a human, a timer, a condition, or
+ /// child branches is wall time, not active time.
+ ///
+ /// Active is clamped to wall. A worker killed mid-turn leaves its bracket
+ /// open forever (see [`StageInferenceProjection`]), and without the clamp
+ /// that bracket would tick up without bound. The clamp does not need to
+ /// know the worker died: a stage cannot have been active longer than it
+ /// has existed. `watchdog.timeout` remains the authority on whether a run
+ /// is actually stuck.
+ #[must_use]
+ pub fn live_timing(&self, now: DateTime) -> StageTiming {
+ if let Some(timing) = self.timing {
+ return timing;
+ }
+
+ let wall_time_ms = self.live_wall_time_ms(now).unwrap_or(0);
+
+ // `handler` is absent on projections built from events written before
+ // stage execution identity. Treat those as agent stages, matching
+ // `StageHandler::from_handler_type`: the accumulators below are only
+ // ever populated by agent events, so a legacy non-agent stage still
+ // reads zero rather than being credited work it never did.
+ let handler = self.handler.unwrap_or(StageHandler::Agent);
+ let (inference_time_ms, tool_time_ms) = match handler {
+ StageHandler::Agent => {
+ let open_inference = self
+ .inference
+ .as_ref()
+ .map(|inference| inference.started_at)
+ .or(self.acp_started_at)
+ .map_or(0, |started_at| timing::elapsed_ms(started_at, now));
+ let open_tool = self
+ .tool_batch
+ .as_ref()
+ .map_or(0, |batch| timing::elapsed_ms(batch.started_at, now));
+ (
+ self.live_inference_ms.saturating_add(open_inference),
+ self.live_tool_ms.saturating_add(open_tool),
+ )
+ }
+ StageHandler::Prompt => (wall_time_ms, 0),
+ StageHandler::Command => (0, wall_time_ms),
+ // Waiting on a human, a timer, a condition, or child branches is
+ // wall time, not active time.
+ StageHandler::Human
+ | StageHandler::Wait
+ | StageHandler::Conditional
+ | StageHandler::Parallel
+ | StageHandler::ParallelFanIn
+ | StageHandler::StackManagerLoop
+ | StageHandler::Start
+ | StageHandler::Exit => (0, 0),
+ };
+
+ StageTiming::new(wall_time_ms, inference_time_ms, tool_time_ms).clamped_to_wall()
+ }
+
+ /// Fold a closed inference bracket into the live accumulator.
+ pub fn accumulate_inference_ms(&mut self, elapsed_ms: u64) {
+ self.live_inference_ms = self.live_inference_ms.saturating_add(elapsed_ms);
+ }
+
+ /// Record the start of an external ACP agent process.
+ pub fn open_acp_inference(&mut self, started_at: DateTime) {
+ self.close_open_acp_inference(started_at);
+ self.acp_started_at = Some(started_at);
+ }
+
+ /// Close an ACP process with its measured duration.
+ ///
+ /// The duration covers the complete process. Use the larger value so a
+ /// replayed terminal event or a projection restored mid-process cannot
+ /// double-count it.
+ pub fn close_acp_inference(&mut self, duration_ms: u64) {
+ self.acp_started_at = None;
+ self.live_inference_ms = self.live_inference_ms.max(duration_ms);
+ }
+
+ /// Fold an open ACP process into the live accumulator at a run boundary.
+ pub fn close_open_acp_inference(&mut self, now: DateTime) {
+ let Some(started_at) = self.acp_started_at.take() else {
+ return;
+ };
+ self.accumulate_inference_ms(timing::elapsed_ms(started_at, now));
+ }
+
+ /// Record a dispatched tool call, opening a batch if none is outstanding.
+ ///
+ /// If a replacement root session starts work before the old session's end
+ /// event arrives, freeze the old batch at this boundary before opening
+ /// the new one. This keeps the sessions separate without dropping time.
+ pub fn open_tool_call(
+ &mut self,
+ session_id: String,
+ tool_call_id: String,
+ started_at: DateTime,
+ ) {
+ let replaces_open_batch = self
+ .tool_batch
+ .as_ref()
+ .is_some_and(|batch| batch.session_id != session_id);
+ if replaces_open_batch {
+ self.close_open_tool_batch(started_at);
+ }
+ self.tool_batch
+ .get_or_insert_with(|| StageToolBatchProjection {
+ session_id,
+ started_at,
+ open_call_ids: BTreeSet::new(),
+ })
+ .open_call_ids
+ .insert(tool_call_id);
+ }
+
+ /// Retire a tool call. Folds the batch into the live accumulator once the
+ /// last outstanding call reports, so concurrent calls count once.
+ pub fn close_tool_call(&mut self, session_id: &str, tool_call_id: &str, now: DateTime) {
+ let Some(batch) = self.tool_batch.as_mut() else {
+ return;
+ };
+ if batch.session_id != session_id
+ || !batch.open_call_ids.remove(tool_call_id)
+ || !batch.open_call_ids.is_empty()
+ {
+ return;
+ }
+ self.close_open_tool_batch(now);
+ }
+
+ /// Close a tool batch only when it belongs to `session_id`.
+ pub fn close_tool_batch_for_session(&mut self, session_id: &str, now: DateTime) {
+ let opened_here = self
+ .tool_batch
+ .as_ref()
+ .is_some_and(|batch| batch.session_id == session_id);
+ if opened_here {
+ self.close_open_tool_batch(now);
+ }
+ }
+
+ /// Fold any open tool batch into the live accumulator.
+ pub fn close_open_tool_batch(&mut self, now: DateTime) {
+ let Some(batch) = self.tool_batch.take() else {
+ return;
+ };
+ self.live_tool_ms = self
+ .live_tool_ms
+ .saturating_add(timing::elapsed_ms(batch.started_at, now));
+ }
+
+ /// Install a worker-provided terminal timing and discard transient live
+ /// bookkeeping that is no longer authoritative.
+ pub fn set_authoritative_timing(&mut self, timing: StageTiming) {
+ self.timing = Some(timing);
+ self.clear_live_timing();
+ }
+
+ /// Discard transient timing accumulators and open brackets.
+ pub fn clear_live_timing(&mut self) {
+ self.live_inference_ms = 0;
+ self.live_tool_ms = 0;
+ self.inference = None;
+ self.acp_started_at = None;
+ self.tool_batch = None;
+ }
+
/// Begin a new automatic attempt within this stage execution: clear every
/// per-attempt field so prior-attempt data does not leak, then record
/// `started_at` and `state = Running`. Preserves `first_event_seq`
@@ -709,18 +974,21 @@ impl RunProjection {
/// terminal conclusion yet.
///
/// Run-level wall time ticks from `run.started` to `now`. Active time sums
- /// inference and tool timing from stages that have already emitted a
- /// terminal stage event. Stage projections do not currently track live
- /// inference/tool time while a stage is still running, so active time steps
- /// forward when each stage completes while wall time advances continuously.
+ /// [`StageProjection::live_timing`] across every stage, so an in-flight
+ /// stage contributes its live estimate rather than nothing — both halves
+ /// advance continuously. Terminal stages contribute their finalized,
+ /// authoritative breakdown.
+ ///
+ /// Active is not clamped to run wall time here: concurrent branches can
+ /// legitimately sum past it. The clamp applies per stage.
#[must_use]
pub fn live_run_timing(&self, now: DateTime) -> Option {
let start = self.start.as_ref()?;
- let wall_time_ms = RunTiming::wall_time_ms_since(start.start_time, now);
+ let wall_time_ms = timing::elapsed_ms(start.start_time, now);
let active = self
.stages
.values()
- .filter_map(|stage| stage.timing)
+ .map(|stage| stage.live_timing(now))
.fold(RunTiming::default(), |acc, timing| {
acc.saturating_add(&RunTiming::from(timing))
});
@@ -844,11 +1112,13 @@ mod iter_stages_tests {
use std::num::NonZeroU32;
use chrono::Utc;
+ use fabro_model::{Catalog, ModelRef, ProviderId};
use serde_json::json;
use super::RunProjection;
use crate::{
- AgentControlState, Graph, RunId, RunSpec, StageProjection, WorkflowSettings, test_support,
+ AgentControlState, BilledTokenCounts, Graph, RunId, RunSpec, StageProjection,
+ WorkflowSettings, test_support,
};
fn seq(n: u32) -> NonZeroU32 {
@@ -970,4 +1240,361 @@ mod iter_stages_tests {
assert_eq!(order, vec!["build@1", "verify@1", "verify@2"]);
}
}
+
+ fn priced_stage(total_usd_micros: Option) -> StageProjection {
+ let mut stage = StageProjection::new(seq(1));
+ stage.usage = BilledTokenCounts {
+ input_tokens: 500_000,
+ output_tokens: 125_000,
+ total_tokens: 625_000,
+ total_usd_micros,
+ ..BilledTokenCounts::default()
+ };
+ stage.model = Some(ModelRef {
+ provider: ProviderId::openai(),
+ model_id: "gpt-5.4".into(),
+ speed: None,
+ });
+ stage
+ }
+
+ #[test]
+ fn billed_usage_prices_uncosted_tokens_from_the_catalog() {
+ let stage = priced_stage(None);
+
+ assert_eq!(stage.billed_usage(None).total_usd_micros, None);
+ let priced = stage.billed_usage(Some(Catalog::builtin()));
+ assert!(
+ priced.total_usd_micros.is_some_and(|cost| cost > 0),
+ "expected a catalog price, got {:?}",
+ priced.total_usd_micros
+ );
+ // Pricing only fills in the cost; the token buckets pass through.
+ assert_eq!(priced.input_tokens, 500_000);
+ assert_eq!(priced.output_tokens, 125_000);
+ }
+
+ #[test]
+ fn billed_usage_keeps_a_provider_reported_cost_over_the_catalog_estimate() {
+ let stage = priced_stage(Some(42));
+
+ assert_eq!(
+ stage
+ .billed_usage(Some(Catalog::builtin()))
+ .total_usd_micros,
+ Some(42)
+ );
+ }
+
+ #[test]
+ fn billed_usage_leaves_a_modelless_stage_uncosted() {
+ let mut stage = priced_stage(None);
+ stage.model = None;
+
+ assert_eq!(
+ stage
+ .billed_usage(Some(Catalog::builtin()))
+ .total_usd_micros,
+ None
+ );
+ }
+
+ #[test]
+ fn billed_usage_leaves_zero_tokens_uncosted() {
+ let mut stage = priced_stage(None);
+ stage.usage = BilledTokenCounts::default();
+
+ assert_eq!(
+ stage
+ .billed_usage(Some(Catalog::builtin()))
+ .total_usd_micros,
+ None
+ );
+ }
+}
+
+#[cfg(test)]
+mod live_timing_tests {
+ use std::collections::HashMap;
+
+ use chrono::{DateTime, TimeZone, Utc};
+
+ use super::{RunProjection, StageToolBatchProjection};
+ use crate::{
+ Graph, ModelRef, RunId, RunSpec, StageHandler, StageInferenceProjection, StageProjection,
+ StageState, StageTiming, StartRecord, WorkflowSettings, first_event_seq, test_support,
+ };
+
+ fn at(seconds: i64) -> DateTime {
+ Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap()
+ }
+
+ fn projection() -> RunProjection {
+ RunProjection::new(
+ "Test run".to_string(),
+ RunSpec {
+ run_id: RunId::new(),
+ settings: WorkflowSettings::default(),
+ graph: Graph::new("test"),
+ graph_source: None,
+ workflow_slug: None,
+ automation: None,
+ source_directory: None,
+ labels: HashMap::default(),
+ provenance: test_support::test_run_provenance(),
+ manifest_blob: None,
+ definition_blob: None,
+ git: None,
+ fork_source_ref: None,
+ },
+ at(0),
+ )
+ }
+
+ /// In-flight stage that started at `at(0)`.
+ fn running(handler: StageHandler) -> StageProjection {
+ let mut stage = StageProjection::new(first_event_seq(1));
+ stage.handler = Some(handler);
+ stage.started_at = Some(at(0));
+ stage.state = StageState::Running;
+ stage
+ }
+
+ fn open_bracket(started_at: DateTime) -> StageInferenceProjection {
+ StageInferenceProjection {
+ session_id: "session-1".to_string(),
+ started_at,
+ requested_model: ModelRef {
+ provider: "anthropic".parse().unwrap(),
+ model_id: "claude-sonnet-5".into(),
+ speed: None,
+ },
+ first_output_at: None,
+ first_output_kind: None,
+ retries: 0,
+ }
+ }
+
+ #[test]
+ fn terminal_stage_returns_stored_timing_unchanged() {
+ let mut stage = running(StageHandler::Agent);
+ stage.state = StageState::Succeeded;
+ stage.timing = Some(StageTiming::new(90_000, 78_000, 7_000));
+ // Live accumulators are stale leftovers; the finalized value wins.
+ stage.live_inference_ms = 5;
+ stage.live_tool_ms = 5;
+
+ assert_eq!(
+ stage.live_timing(at(600)),
+ StageTiming::new(90_000, 78_000, 7_000)
+ );
+ }
+
+ #[test]
+ fn agent_stage_sums_accumulators_and_open_brackets() {
+ let mut stage = running(StageHandler::Agent);
+ stage.live_inference_ms = 30_000;
+ stage.live_tool_ms = 5_000;
+ // Inference open for 20s, tools open for 10s, at t=120s.
+ stage.inference = Some(open_bracket(at(100)));
+ stage.tool_batch = Some(StageToolBatchProjection {
+ session_id: "session-1".to_string(),
+ started_at: at(110),
+ open_call_ids: ["call-1".to_string()].into_iter().collect(),
+ });
+
+ assert_eq!(
+ stage.live_timing(at(120)),
+ StageTiming::new(120_000, 50_000, 15_000)
+ );
+ }
+
+ #[test]
+ fn agent_stage_without_brackets_reports_only_accumulators() {
+ let mut stage = running(StageHandler::Agent);
+ stage.live_inference_ms = 30_000;
+ stage.live_tool_ms = 5_000;
+
+ assert_eq!(
+ stage.live_timing(at(120)),
+ StageTiming::new(120_000, 30_000, 5_000)
+ );
+ }
+
+ #[test]
+ fn acp_process_counts_as_live_inference_and_uses_measured_duration() {
+ let mut stage = running(StageHandler::Agent);
+ stage.open_acp_inference(at(30));
+
+ assert_eq!(
+ stage.live_timing(at(45)),
+ StageTiming::new(45_000, 15_000, 0)
+ );
+
+ stage.close_acp_inference(14_500);
+ assert_eq!(
+ stage.live_timing(at(45)),
+ StageTiming::new(45_000, 14_500, 0)
+ );
+ }
+
+ #[test]
+ fn prompt_stage_counts_elapsed_as_inference() {
+ let stage = running(StageHandler::Prompt);
+
+ assert_eq!(
+ stage.live_timing(at(45)),
+ StageTiming::new(45_000, 45_000, 0)
+ );
+ }
+
+ #[test]
+ fn command_stage_counts_elapsed_as_tool() {
+ let stage = running(StageHandler::Command);
+
+ assert_eq!(
+ stage.live_timing(at(45)),
+ StageTiming::new(45_000, 0, 45_000)
+ );
+ }
+
+ #[test]
+ fn waiting_handlers_report_wall_time_with_zero_active() {
+ for handler in [
+ StageHandler::Human,
+ StageHandler::Wait,
+ StageHandler::Conditional,
+ StageHandler::Parallel,
+ StageHandler::ParallelFanIn,
+ StageHandler::StackManagerLoop,
+ StageHandler::Start,
+ StageHandler::Exit,
+ ] {
+ let stage = running(handler);
+ let timing = stage.live_timing(at(600));
+
+ assert_eq!(
+ timing,
+ StageTiming::new(600_000, 0, 0),
+ "{handler} should report wall time only"
+ );
+ }
+ }
+
+ #[test]
+ fn open_bracket_from_a_killed_worker_is_clamped_to_wall() {
+ let mut stage = running(StageHandler::Agent);
+ // Bracket opened before the stage even started — the pathological
+ // shape a killed worker leaves behind. Without the clamp this would
+ // report 700s of inference against 600s of wall.
+ stage.inference = Some(open_bracket(at(-100)));
+
+ let timing = stage.live_timing(at(600));
+
+ assert_eq!(timing.wall_time_ms, 600_000);
+ assert_eq!(timing.active_time_ms, 600_000);
+ }
+
+ #[test]
+ fn clamping_preserves_the_inference_tool_split() {
+ let mut stage = running(StageHandler::Agent);
+ // 3:1 inference:tool, totalling 200s of active against 100s of wall.
+ stage.live_inference_ms = 150_000;
+ stage.live_tool_ms = 50_000;
+
+ let timing = stage.live_timing(at(100));
+
+ assert_eq!(timing.wall_time_ms, 100_000);
+ assert_eq!(timing.active_time_ms, 100_000);
+ assert_eq!(timing.inference_time_ms, 75_000);
+ assert_eq!(timing.tool_time_ms, 25_000);
+ }
+
+ #[test]
+ fn live_run_timing_counts_in_flight_stages_not_just_terminal_ones() {
+ // The shape that motivated this change: two finished stages and one
+ // long-running agent stage that had been active nearly the whole run.
+ let mut projection = projection();
+ projection.start = Some(StartRecord {
+ start_time: at(0),
+ run_branch: None,
+ base_sha: None,
+ });
+
+ let baseline = projection.stage_entry("baseline", 1, first_event_seq(1));
+ baseline.handler = Some(StageHandler::Command);
+ baseline.state = StageState::Succeeded;
+ baseline.timing = Some(StageTiming::new(42_666, 0, 42_663));
+
+ let assess = projection.stage_entry("assess", 1, first_event_seq(2));
+ assess.handler = Some(StageHandler::Agent);
+ assess.state = StageState::Succeeded;
+ assess.timing = Some(StageTiming::new(86_025, 78_230, 7_588));
+
+ let plan = projection.stage_entry("plan", 1, first_event_seq(3));
+ plan.handler = Some(StageHandler::Agent);
+ plan.started_at = Some(at(146));
+ plan.state = StageState::Running;
+ plan.live_inference_ms = 700_000;
+ plan.live_tool_ms = 150_000;
+
+ let timing = projection.live_run_timing(at(1_013)).unwrap();
+
+ assert_eq!(timing.wall_time_ms, 1_013_000);
+ assert_eq!(timing.inference_time_ms, 778_230);
+ assert_eq!(timing.tool_time_ms, 200_251);
+ // Before this change the in-flight stage contributed nothing and the
+ // run reported 128,481 ms of active time against 1,013,000 ms of wall.
+ assert_eq!(timing.active_time_ms, 978_481);
+ }
+
+ #[test]
+ fn live_run_timing_may_exceed_run_wall_when_branches_overlap() {
+ let mut projection = projection();
+ projection.start = Some(StartRecord {
+ start_time: at(0),
+ run_branch: None,
+ base_sha: None,
+ });
+
+ for (index, node) in ["branch-a", "branch-b", "branch-c"].iter().enumerate() {
+ let stage =
+ projection.stage_entry(node, 1, first_event_seq(u32::try_from(index).unwrap() + 1));
+ stage.handler = Some(StageHandler::Agent);
+ stage.state = StageState::Succeeded;
+ stage.timing = Some(StageTiming::new(60_000, 60_000, 0));
+ }
+
+ let timing = projection.live_run_timing(at(60)).unwrap();
+
+ assert_eq!(timing.wall_time_ms, 60_000);
+ assert_eq!(
+ timing.active_time_ms, 180_000,
+ "concurrent branches legitimately sum past run wall time"
+ );
+ }
+
+ #[test]
+ fn a_legacy_stage_without_a_recorded_handler_uses_its_accumulators() {
+ let mut stage = StageProjection::new(first_event_seq(1));
+ stage.handler = None;
+ stage.started_at = Some(at(0));
+ stage.state = StageState::Running;
+ stage.live_inference_ms = 30_000;
+
+ assert_eq!(
+ stage.live_timing(at(120)),
+ StageTiming::new(120_000, 30_000, 0)
+ );
+ }
+
+ #[test]
+ fn a_legacy_stage_with_no_accumulators_reports_no_active_time() {
+ let mut stage = StageProjection::new(first_event_seq(1));
+ stage.handler = None;
+ stage.started_at = Some(at(0));
+ stage.state = StageState::Running;
+
+ assert_eq!(stage.live_timing(at(120)), StageTiming::new(120_000, 0, 0));
+ }
}
diff --git a/lib/foundation/fabro-types/src/timing.rs b/lib/foundation/fabro-types/src/timing.rs
index 0d1e555e8..c7a03d1e9 100644
--- a/lib/foundation/fabro-types/src/timing.rs
+++ b/lib/foundation/fabro-types/src/timing.rs
@@ -18,6 +18,16 @@
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
+/// Non-negative milliseconds between two instants.
+///
+/// Clock skew or an out-of-order replay can put `end` before `start`; those
+/// spans contribute zero rather than wrapping into a large unsigned value.
+#[must_use]
+pub fn elapsed_ms(start: DateTime, end: DateTime) -> u64 {
+ u64::try_from(end.signed_duration_since(start).num_milliseconds().max(0))
+ .expect("non-negative chrono millisecond durations fit in u64")
+}
+
/// Timing breakdown for one stage visit.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct StageTiming {
@@ -62,6 +72,34 @@ impl StageTiming {
Self::new(0, inference_time_ms, tool_time_ms)
}
+ /// Scale the breakdown down so `active_time_ms` does not exceed
+ /// `wall_time_ms`, preserving the inference/tool ratio.
+ ///
+ /// Only meaningful for live estimates of a single in-flight stage, where
+ /// an open bracket left behind by a killed worker would otherwise tick up
+ /// without bound. Finalized timings come from the worker's stopwatch and
+ /// already satisfy the invariant.
+ ///
+ /// Deliberately *not* applied at run level: concurrent branches can
+ /// legitimately sum past run wall time.
+ #[must_use]
+ pub fn clamped_to_wall(&self) -> Self {
+ let active_time_ms = u128::from(self.inference_time_ms) + u128::from(self.tool_time_ms);
+ if active_time_ms <= u128::from(self.wall_time_ms) {
+ return *self;
+ }
+ // Preserve the split rather than truncating one side, so a clamped
+ // stage still shows where its time went. Widen for the multiply: the
+ // quotient is bounded by `wall_time_ms` because the exact, widened
+ // active total exceeds it here, so it always fits back into u64.
+ let scaled =
+ u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) / active_time_ms;
+ let inference_time_ms =
+ u64::try_from(scaled).expect("scaled inference time is bounded by wall time");
+ let tool_time_ms = self.wall_time_ms.saturating_sub(inference_time_ms);
+ Self::new(self.wall_time_ms, inference_time_ms, tool_time_ms)
+ }
+
/// Sum two timings field-by-field. Used to aggregate visits of one node
/// and to accumulate run-level rollups.
#[must_use]
@@ -136,13 +174,6 @@ impl RunTiming {
..self
}
}
-
- /// Milliseconds elapsed from `start` to `now`, clamped at zero.
- #[must_use]
- pub fn wall_time_ms_since(start: DateTime, now: DateTime) -> u64 {
- u64::try_from(now.signed_duration_since(start).num_milliseconds().max(0))
- .expect("non-negative milliseconds fit in u64")
- }
}
impl From for RunTiming {
@@ -158,7 +189,9 @@ impl From for RunTiming {
#[cfg(test)]
mod tests {
- use super::{RunTiming, StageTiming};
+ use chrono::{TimeZone, Utc};
+
+ use super::{RunTiming, StageTiming, elapsed_ms};
#[test]
fn stage_timing_new_derives_active_as_sum_of_inference_and_tool() {
@@ -189,6 +222,23 @@ mod tests {
assert_eq!(sum.active_time_ms, 175);
}
+ #[test]
+ fn stage_timing_clamp_uses_the_unsaturated_active_total() {
+ let timing = StageTiming::new(u64::MAX, u64::MAX, u64::MAX).clamped_to_wall();
+
+ assert_eq!(timing.inference_time_ms, u64::MAX / 2);
+ assert_eq!(timing.tool_time_ms, u64::MAX.saturating_sub(u64::MAX / 2));
+ assert_eq!(timing.active_time_ms, u64::MAX);
+ }
+
+ #[test]
+ fn elapsed_ms_clamps_out_of_order_instants_to_zero() {
+ let later = Utc.timestamp_opt(100, 0).unwrap();
+ let earlier = Utc.timestamp_opt(99, 0).unwrap();
+
+ assert_eq!(elapsed_ms(later, earlier), 0);
+ }
+
#[test]
fn run_timing_wall_only_zeroes_breakdown_and_active() {
let timing = RunTiming::wall_only(1500);
diff --git a/lib/packages/fabro-api-client/scripts/normalize-generated.ts b/lib/packages/fabro-api-client/scripts/normalize-generated.ts
index 4bd6f85b4..aa5dbd81e 100644
--- a/lib/packages/fabro-api-client/scripts/normalize-generated.ts
+++ b/lib/packages/fabro-api-client/scripts/normalize-generated.ts
@@ -4,6 +4,10 @@
// blank lines at end of file. Left alone, every regeneration produces a noisy
// whitespace diff that masks real spec/client drift. This pass strips trailing
// whitespace from every line and ends each file with exactly one newline.
+//
+// openapi-generator maps arrays with `uniqueItems` to `Set`, but Axios
+// decodes and encodes their JSON representation as arrays. Keep the generated
+// type aligned with the runtime wire value.
import { Glob } from "bun";
@@ -14,6 +18,7 @@ for await (const path of glob.scan(".")) {
const original = await Bun.file(path).text();
const normalized =
original
+ .replace(/\bSet line.replace(/\s+$/, ""))
.join("\n")
diff --git a/lib/packages/fabro-api-client/src/.openapi-generator/FILES b/lib/packages/fabro-api-client/src/.openapi-generator/FILES
index e3a1a7c0e..4ae265b76 100644
--- a/lib/packages/fabro-api-client/src/.openapi-generator/FILES
+++ b/lib/packages/fabro-api-client/src/.openapi-generator/FILES
@@ -487,6 +487,7 @@ models/stage-projection.ts
models/stage-state.ts
models/stage-summary.ts
models/stage-timing.ts
+models/stage-tool-batch-projection.ts
models/start-record.ts
models/start-run-request.ts
models/steer-run-request.ts
diff --git a/lib/packages/fabro-api-client/src/models/batch-delete-runs-request.ts b/lib/packages/fabro-api-client/src/models/batch-delete-runs-request.ts
index 938a9514e..039b5ffbc 100644
--- a/lib/packages/fabro-api-client/src/models/batch-delete-runs-request.ts
+++ b/lib/packages/fabro-api-client/src/models/batch-delete-runs-request.ts
@@ -21,7 +21,7 @@ export interface BatchDeleteRunsRequest {
/**
* Run IDs to process, in result order.
*/
- 'run_ids': Set;
+ 'run_ids': Array;
/**
* Whether to force deletion of active runs. Defaults to `false`.
*/
diff --git a/lib/packages/fabro-api-client/src/models/batch-run-lifecycle-request.ts b/lib/packages/fabro-api-client/src/models/batch-run-lifecycle-request.ts
index b9225f933..d93be0250 100644
--- a/lib/packages/fabro-api-client/src/models/batch-run-lifecycle-request.ts
+++ b/lib/packages/fabro-api-client/src/models/batch-run-lifecycle-request.ts
@@ -21,5 +21,5 @@ export interface BatchRunLifecycleRequest {
/**
* Run IDs to process, in result order.
*/
- 'run_ids': Set;
+ 'run_ids': Array;
}
diff --git a/lib/packages/fabro-api-client/src/models/index.ts b/lib/packages/fabro-api-client/src/models/index.ts
index 5cd29601d..029a2d3ab 100644
--- a/lib/packages/fabro-api-client/src/models/index.ts
+++ b/lib/packages/fabro-api-client/src/models/index.ts
@@ -457,6 +457,7 @@ export * from './stage-projection';
export * from './stage-state';
export * from './stage-summary';
export * from './stage-timing';
+export * from './stage-tool-batch-projection';
export * from './start-record';
export * from './start-run-request';
export * from './steer-run-request';
diff --git a/lib/packages/fabro-api-client/src/models/run-stage.ts b/lib/packages/fabro-api-client/src/models/run-stage.ts
index aafd25f92..ce184f7e8 100644
--- a/lib/packages/fabro-api-client/src/models/run-stage.ts
+++ b/lib/packages/fabro-api-client/src/models/run-stage.ts
@@ -13,6 +13,9 @@
*/
+// May contain unused imports in some cases
+// @ts-ignore
+import type { BilledTokenCounts } from './billed-token-counts';
// May contain unused imports in some cases
// @ts-ignore
import type { StageHandler } from './stage-handler';
@@ -62,4 +65,8 @@ export interface RunStage {
* Wall-clock time the latest attempt of this stage started, if known.
*/
'started_at'?: string | null;
+ /**
+ * Token counts for this stage execution alone. `total_usd_micros` is the provider-reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/billing` rows use. All-zero counts mean the stage made no model calls. Unlike the billing rows, which sum every visit of a node, this covers only this visit.
+ */
+ 'billing': BilledTokenCounts;
}
diff --git a/lib/packages/fabro-api-client/src/models/run-timing.ts b/lib/packages/fabro-api-client/src/models/run-timing.ts
index fb585b0e9..f0344e5d3 100644
--- a/lib/packages/fabro-api-client/src/models/run-timing.ts
+++ b/lib/packages/fabro-api-client/src/models/run-timing.ts
@@ -15,12 +15,12 @@
/**
- * Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently.
+ * Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. For a running run, stages still in flight contribute a live estimate rather than nothing, so wall and active both advance continuously. Unlike `StageTiming`, active is not clamped to wall here — concurrent branches can legitimately sum past run wall time.
*/
export interface RunTiming {
'wall_time_ms': number;
- 'inference_time_ms'?: number;
- 'tool_time_ms'?: number;
+ 'inference_time_ms': number;
+ 'tool_time_ms': number;
/**
* Equals `inference_time_ms + tool_time_ms`.
*/
diff --git a/lib/packages/fabro-api-client/src/models/stage-projection.ts b/lib/packages/fabro-api-client/src/models/stage-projection.ts
index 57fd342ca..032835c95 100644
--- a/lib/packages/fabro-api-client/src/models/stage-projection.ts
+++ b/lib/packages/fabro-api-client/src/models/stage-projection.ts
@@ -60,6 +60,9 @@ import type { StageState } from './stage-state';
import type { StageTiming } from './stage-timing';
// May contain unused imports in some cases
// @ts-ignore
+import type { StageToolBatchProjection } from './stage-tool-batch-projection';
+// May contain unused imports in some cases
+// @ts-ignore
import type { SubAgentProjection } from './sub-agent-projection';
// May contain unused imports in some cases
// @ts-ignore
@@ -96,6 +99,15 @@ export interface StageProjection {
*/
'started_at'?: string | null;
'timing'?: StageTiming | null;
+ /**
+ * Inference time accumulated from closed brackets during the current attempt. Live estimate only — the authoritative value arrives with the terminal event and lands in `timing`. Excludes the currently open bracket, whose span is measured from `inference.started_at`.
+ */
+ 'live_inference_ms'?: number;
+ /**
+ * Tool time accumulated from closed tool batches during the current attempt. A batch spans the first dispatched call through the completion that drains the last outstanding one, so tools running concurrently within a turn are counted once.
+ */
+ 'live_tool_ms'?: number;
+ 'tool_batch'?: StageToolBatchProjection | null;
'usage': BilledTokenCounts;
'model'?: BillingModelRef | null;
'todos'?: TodoListProjection | null;
@@ -118,6 +130,10 @@ export interface StageProjection {
'mcp_servers'?: Array;
'context_window'?: StageContextWindowProjection | null;
'inference'?: StageInferenceProjection | null;
+ /**
+ * Start of an external ACP agent process, if one is running. ACP agents do not expose Fabro\'s internal LLM brackets, so the process lifetime supplies their live inference estimate.
+ */
+ 'acp_started_at'?: string | null;
/**
* Whether the agent is executing normally or waiting for steering after an interrupt.
*/
diff --git a/lib/packages/fabro-api-client/src/models/stage-timing.ts b/lib/packages/fabro-api-client/src/models/stage-timing.ts
index d9915c409..4b106816c 100644
--- a/lib/packages/fabro-api-client/src/models/stage-timing.ts
+++ b/lib/packages/fabro-api-client/src/models/stage-timing.ts
@@ -15,12 +15,12 @@
/**
- * Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`.
+ * Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. For a terminal stage these come from the worker\'s own stopwatch and are authoritative. For a stage still in flight they are a live estimate reconstructed from the event log, and `active_time_ms` is clamped to `wall_time_ms`. The estimate is replaced by the authoritative breakdown when the stage reaches a terminal event.
*/
export interface StageTiming {
'wall_time_ms': number;
- 'inference_time_ms'?: number;
- 'tool_time_ms'?: number;
+ 'inference_time_ms': number;
+ 'tool_time_ms': number;
/**
* Equals `inference_time_ms + tool_time_ms`.
*/
diff --git a/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts
new file mode 100644
index 000000000..971fa7c59
--- /dev/null
+++ b/lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts
@@ -0,0 +1,33 @@
+/* tslint:disable */
+/* eslint-disable */
+/**
+ * Fabro Run API
+ * HTTP API for managing Fabro workflow run executions.
+ *
+ * The version of the OpenAPI document: 0.1.0
+ *
+ *
+ * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
+ * https://openapi-generator.tech
+ * Do not edit the class manually.
+ */
+
+
+
+/**
+ * One open tool batch: tool calls dispatched together that have not all reported completion. `open_call_ids` is a set rather than a count so a duplicated completion in a replayed log cannot drain the batch early.
+ */
+export interface StageToolBatchProjection {
+ /**
+ * Root agent session that dispatched the batch. Transitions are gated on it so delayed events from a replaced session cannot mutate the current batch.
+ */
+ 'session_id': string;
+ /**
+ * When the batch opened — the first dispatched call observed while no other calls were outstanding.
+ */
+ 'started_at': string;
+ /**
+ * Calls dispatched but not yet completed, by tool call id.
+ */
+ 'open_call_ids': Array;
+}
diff --git a/test/edge_only_node.fabro b/test/edge_only_node.fabro
new file mode 100644
index 000000000..3d45b3346
--- /dev/null
+++ b/test/edge_only_node.fabro
@@ -0,0 +1,10 @@
+digraph EdgeOnlyNode {
+ graph [goal="Reference a node that was never declared"]
+
+ /* `misspelled_node` is only ever named by an edge, never declared. */
+ start [shape=Mdiamond, label="Start"]
+ exit [shape=Msquare, label="Exit"]
+
+ start -> misspelled_node
+ misspelled_node -> exit
+}
diff --git a/test/server-model.fabro b/test/server-model.fabro
new file mode 100644
index 000000000..099f3c8d9
--- /dev/null
+++ b/test/server-model.fabro
@@ -0,0 +1,10 @@
+digraph ServerModel {
+ graph [goal="Use a server-owned model"]
+
+ start [shape=Mdiamond, label="Start"]
+ exit [shape=Msquare, label="Exit"]
+
+ work [label="Work", prompt="Do work", model="private-model", provider="server-only"]
+
+ start -> work -> exit
+}