Merge remote-tracking branch 'origin/main' into codex/pr669-merge-main-3d8cf48

# Conflicts:
#	apps/fabro-web/app/routes/run-detail/model.ts
This commit is contained in:
Bryan Helmkamp 2026-07-28 15:35:42 -04:00
commit 96604429af
No known key found for this signature in database
88 changed files with 3803 additions and 1046 deletions

View file

@ -11,10 +11,6 @@ leak-timeout = "500ms"
filter = "package(fabro-server)"
slow-timeout = { period = "5s", terminate-after = 4 }
[[profile.default.overrides]]
filter = "package(fabro-server) & test(all_spec_routes_are_routable)"
slow-timeout = { period = "15s", terminate-after = 4 }
[[profile.default.overrides]]
filter = "package(fabro-workflow)"
slow-timeout = { period = "2s", terminate-after = 3 }

View file

@ -115,8 +115,10 @@ export function RunTableRow({
</td>
)}
{show("size") && (
<td className="whitespace-nowrap px-3 py-2.5 text-center">
{run.size != null && <SizeChip size={run.size} />}
<td className="relative z-10 px-3 py-2.5 text-center whitespace-nowrap">
{run.size != null && (
<SizeChip size={run.size} totalUsdMicros={run.totalUsdMicros} />
)}
</td>
)}
{show("changes") && (

View file

@ -0,0 +1,58 @@
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import TestRenderer, { act } from "react-test-renderer";
import { setupReactTestEnv } from "../lib/test-utils";
import { SizeChip } from "./size-chip";
import { Tooltip } from "./ui";
let teardownReactTestEnv: (() => void) | undefined;
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
function render(element: React.ReactElement): TestRenderer.ReactTestRenderer {
let renderer: TestRenderer.ReactTestRenderer | undefined;
act(() => {
renderer = TestRenderer.create(element);
});
mountedRenderers.push(renderer!);
return renderer!;
}
function tooltipLabel(element: React.ReactElement): string {
return render(element).root.findByType(Tooltip).props.label as string;
}
describe("SizeChip", () => {
beforeEach(() => {
teardownReactTestEnv = setupReactTestEnv();
});
afterEach(() => {
act(() => {
for (const renderer of mountedRenderers.splice(0)) {
renderer.unmount();
}
});
teardownReactTestEnv?.();
teardownReactTestEnv = undefined;
});
test("renders the size letter", () => {
expect(JSON.stringify(render(<SizeChip size="M" />).toJSON())).toContain("M");
});
test("appends the cost to the tooltip", () => {
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={12_340_000} />))
.toBe("Size M · $12.34");
});
test("omits the cost when the run has no billing yet", () => {
expect(tooltipLabel(<SizeChip size="M" />)).toBe("Size M");
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={null} />)).toBe("Size M");
});
test("calls out the tiers that warrant attention", () => {
expect(tooltipLabel(<SizeChip size="L" totalUsdMicros={150_000_000} />))
.toBe("Size L (risky) · $150.00");
expect(tooltipLabel(<SizeChip size="XL" />)).toBe("Size XL (unhealthy)");
});
});

View file

@ -1,3 +1,4 @@
import { memo } from "react";
import type { RunSize } from "@qltysh/fabro-api-client";
import { formatUsdMicros } from "../lib/format";
@ -11,7 +12,7 @@ const SIZE_TONE: Record<RunSize, { className: string; note: string | null }> = {
XL: { className: "bg-coral/15 text-coral", note: "unhealthy" },
};
export function SizeChip({
export const SizeChip = memo(function SizeChip({
size,
totalUsdMicros,
}: {
@ -19,10 +20,10 @@ export function SizeChip({
totalUsdMicros?: number | null;
}) {
const tone = SIZE_TONE[size];
const billed = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)} billed` : "";
const amount = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)}` : "";
const tooltip = tone.note != null
? `Size ${size} (${tone.note})${billed}`
: `Size ${size}${billed}`;
? `Size ${size} (${tone.note})${amount}`
: `Size ${size}${amount}`;
return (
<Tooltip label={tooltip}>
<span className={`rounded px-1.5 py-0.5 font-mono text-xs font-bold tabular-nums ${tone.className}`}>
@ -30,4 +31,4 @@ export function SizeChip({
</span>
</Tooltip>
);
}
});

View file

@ -9,6 +9,7 @@ import { StagePopover } from "./stage-popover";
import { deriveStageSummary } from "./stage-popover-summary";
import type { Stage } from "../lib/stage-sidebar";
import { generatedAxios } from "../lib/api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
function makeEvent(overrides: Partial<EventEnvelope>): EventEnvelope {
return {
@ -34,6 +35,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
duration: "1m 30s",
startedAt: "2026-05-24T11:58:30Z",
providerUsed: { mode: "policy", model: "claude-opus-4-7", reasoning_effort: "high" },
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -3,6 +3,7 @@ import type { EventEnvelope } from "@qltysh/fabro-api-client";
import TestRenderer, { act } from "react-test-renderer";
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
import type { Stage } from "../stage-sidebar";
import { FanInResults } from "./fan-in-results";
@ -22,6 +23,7 @@ const fanInStage: Stage = {
visit: 1,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
};
function event(seq: number, partial: Partial<EventEnvelope>): EventEnvelope {

View file

@ -4,6 +4,7 @@ import TestRenderer, { act } from "react-test-renderer";
import { MemoryRouter } from "react-router";
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
import type { Stage } from "../stage-sidebar";
import { ParallelChildren } from "./parallel-children";
@ -23,6 +24,7 @@ const parallelStage: Stage = {
visit: 1,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
};
function event(partial: Partial<EventEnvelope>): EventEnvelope {

View file

@ -2,6 +2,7 @@ import { describe, expect, test } from "bun:test";
import TestRenderer, { act } from "react-test-renderer";
import { MemoryRouter } from "react-router";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { StageSidebar, type Stage } from "./stage-sidebar";
function makeStage(overrides: Partial<Stage> = {}): Stage {
@ -17,6 +18,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -103,6 +103,17 @@ describe("mapRunListItem", () => {
expect(mapRunListItem(summary).title).toBe("Untitled run");
});
test("carries the billed total so the size chip can show it on hover", () => {
expect(mapRunListItem(makeRun()).totalUsdMicros).toBe(500000);
});
test("leaves the billed total undefined for runs without terminal billing", () => {
expect(mapRunListItem(makeRun({ billing: null })).totalUsdMicros).toBeUndefined();
expect(
mapRunListItem(makeRun({ billing: { total_usd_micros: null } })).totalUsdMicros,
).toBeUndefined();
});
});
describe("mapRunToRunItem", () => {

View file

@ -46,6 +46,7 @@ export interface RunItem {
createdBy: Principal;
lastEventAt?: string;
size?: RunSize;
totalUsdMicros?: number;
}
export const columnStatuses = [
@ -119,6 +120,7 @@ export function mapRunListItem(item: Run): RunItem {
additions: item.diff?.additions,
deletions: item.diff?.deletions,
size: item.size,
totalUsdMicros: item.billing?.total_usd_micros ?? undefined,
};
}

View file

@ -0,0 +1,28 @@
import type { BilledTokenCounts } from "@qltysh/fabro-api-client";
export interface BillingTokenBucket {
label: string;
value: number;
}
export function billableOutputTokens(billing: BilledTokenCounts): number {
return billing.output_tokens + billing.reasoning_tokens;
}
/** The disjoint token buckets shown in every billing breakdown. */
export function billingTokenBuckets(billing: BilledTokenCounts): BillingTokenBucket[] {
return [
{ label: "Cache read", value: billing.cache_read_tokens },
{ label: "Cache creation", value: billing.cache_write_tokens },
{ label: "Uncached", value: billing.input_tokens },
{ label: "Output", value: billableOutputTokens(billing) },
];
}
export function hasBillingUsage(billing: BilledTokenCounts): boolean {
return (
billing.total_tokens !== 0 ||
(billing.total_usd_micros ?? 0) !== 0 ||
billingTokenBuckets(billing).some((bucket) => bucket.value !== 0)
);
}

View file

@ -1,3 +1,4 @@
import { useCallback } from "react";
import useSWR, { type SWRConfiguration } from "swr";
import type {
ApiQuestion,
@ -73,6 +74,7 @@ import {
type RunFileSelection,
type RunGraphDirection,
} from "./query-keys";
import { isTerminalRunStatus } from "./run-actions";
const immutableOptions: SWRConfiguration = {
revalidateIfStale: false,
@ -179,10 +181,21 @@ export function useRunsPage(opts: RunsPageOptions = {}, enabled = true) {
);
}
export function useRun(id: string | undefined) {
export function useRun(id: string | undefined, refreshInterval?: number) {
const pollingInterval = useCallback(
(run: Run | null | undefined) =>
refreshInterval &&
run?.timestamps.started_at &&
!isTerminalRunStatus(run.lifecycle.status.kind)
? refreshInterval
: 0,
[refreshInterval],
);
return useSWR<Run | null>(
id ? queryKeys.runs.detail(id) : null,
() => apiNullableData(() => runsApi.retrieveRun(id!)),
refreshInterval ? { refreshInterval: pollingInterval } : undefined,
);
}

View file

@ -1,7 +1,6 @@
import { describe, expect, test } from "bun:test";
import { queryKeys } from "./query-keys";
import { queryKeysForRunEvent } from "./run-events";
describe("queryKeys", () => {
test("uses semantic tuples as stable SWR keys and keeps SSE URLs explicit", () => {
@ -61,52 +60,4 @@ describe("queryKeys", () => {
expect(queryKeys.runs.attachUrl("run 1")).toBe("/api/v1/runs/run%201/attach");
});
test("event-mapped keys match query hook resources", () => {
expect(queryKeysForRunEvent("run-1", "checkpoint.completed")).toEqual(
[
...queryKeys.runs.filesAllScopes("run-1"),
queryKeys.runs.commits("run-1"),
],
);
expect(queryKeysForRunEvent("run-1", "stage.completed", "stage-1")).toEqual([
queryKeys.runs.stages("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.graph("run-1", "LR"),
queryKeys.runs.graph("run-1", "TB"),
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
expect(queryKeysForRunEvent("run-1", "run.title.updated")).toEqual([
queryKeys.runs.detail("run-1"),
]);
});
test("agent activity events invalidate per-stage resources", () => {
for (const event of [
"stage.prompt",
"agent.tool.started",
"agent.tool.completed",
"command.started",
"command.completed",
]) {
expect(queryKeysForRunEvent("run-1", event, "stage-1")).toEqual([
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
}
expect(queryKeysForRunEvent("run-1", "agent.message", "stage-1")).toEqual([
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
});
test("agent message without a node_id still invalidates projected state", () => {
expect(queryKeysForRunEvent("run-1", "agent.message")).toEqual([
queryKeys.runs.state("run-1"),
]);
});
});

View file

@ -111,10 +111,7 @@ export async function deleteRuns(
request?: Request,
): Promise<BatchDeleteRunsResponse> {
try {
// See `batchRunLifecycleAction` for the `as unknown as` rationale:
// openapi-generator types `uniqueItems` arrays as `Set<T>` while the wire
// contract is a JSON array.
const body = { run_ids: runIds, force } as unknown as BatchDeleteRunsRequest;
const body: BatchDeleteRunsRequest = { run_ids: runIds, force };
return await apiData(() => runsApi.batchDeleteRuns(body, requestSignalOptions(request)));
} catch (error) {
throw lifecycleActionErrorFromError(error);
@ -284,10 +281,7 @@ async function batchRunLifecycleAction(
request?: Request,
): Promise<BatchRunLifecycleResponse> {
try {
// openapi-generator's TypeScript client represents `uniqueItems` arrays as
// Set<T>, but the HTTP wire contract is still a JSON array. Keep an array
// here so Axios serializes the request body correctly.
const body = { run_ids: runIds } as unknown as BatchRunLifecycleRequest;
const body: BatchRunLifecycleRequest = { run_ids: runIds };
switch (action) {
case "archive":
return await apiData(() => runsApi.batchArchiveRuns(body, requestSignalOptions(request)));

View file

@ -85,6 +85,8 @@ describe("queryKeysForRunEvent", () => {
test("interrupt settlement invalidates projected control state and stage activity", () => {
expect(queryKeysForRunEvent("run-1", "agent.round.interrupted", "nap@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.stageEvents("run-1", "nap@1"),
@ -129,10 +131,16 @@ describe("queryKeysForRunEvent", () => {
test("every inference projection transition invalidates live run state", () => {
for (const event of [
"agent.llm.started",
"agent.llm.first_output",
"agent.llm.retry",
"agent.error",
]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
for (const event of ["agent.llm.first_output", "agent.llm.retry"]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
@ -141,15 +149,47 @@ describe("queryKeysForRunEvent", () => {
expect(
queryKeysForRunEvent("run-1", "agent.message", "code@1"),
).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
expect(queryKeysForRunEvent("run-1", "agent.session.ended")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
]);
});
test("ACP timing events invalidate live summaries and stage events", () => {
for (const event of [
"agent.acp.started",
"agent.acp.completed",
"agent.acp.cancelled",
"agent.acp.timed_out",
]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
});
test("tool timing events invalidate live summaries and stage resources", () => {
for (const event of ["agent.tool.started", "agent.tool.completed"]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
}
});
test("watchdog timeout refreshes the stage events for that stage", () => {
expect(
queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"),

View file

@ -118,6 +118,22 @@ const INFERENCE_EVENTS = new Set([
"agent.error",
"agent.session.ended",
]);
const INFERENCE_TIMING_EVENTS = new Set([
"agent.llm.started",
"agent.message",
"agent.error",
"agent.session.ended",
]);
const TOOL_TIMING_EVENTS = new Set([
"agent.tool.started",
"agent.tool.completed",
]);
const ACP_TIMING_EVENTS = new Set([
"agent.acp.started",
"agent.acp.completed",
"agent.acp.cancelled",
"agent.acp.timed_out",
]);
// Todo / task mutation events refresh `getRunState` consumers (so per-stage
// todo projections update live) and the run events list.
const TODO_EVENTS = new Set([
@ -126,6 +142,14 @@ const TODO_EVENTS = new Set([
"todo.deleted",
]);
function liveTimingKeys(runId: string): Key[] {
return [
queryKeys.runs.detail(runId),
queryKeys.runs.state(runId),
queryKeys.runs.billing(runId),
];
}
export function queryKeysForRunEvent(
runId: string,
event: string,
@ -184,6 +208,12 @@ export function queryKeysForRunEvent(
if (AGENT_CONTROL_STATE_EVENTS.has(event)) {
keys.unshift(queryKeys.runs.state(runId));
}
if (event === "agent.round.interrupted") {
keys.unshift(
queryKeys.runs.detail(runId),
queryKeys.runs.billing(runId),
);
}
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
@ -192,7 +222,9 @@ export function queryKeysForRunEvent(
}
if (INFERENCE_EVENTS.has(event)) {
const keys: Key[] = [queryKeys.runs.state(runId)];
const keys = INFERENCE_TIMING_EVENTS.has(event)
? liveTimingKeys(runId)
: [queryKeys.runs.state(runId)];
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
if (event === "agent.message") {
@ -202,6 +234,23 @@ export function queryKeysForRunEvent(
return keys;
}
if (TOOL_TIMING_EVENTS.has(event)) {
const keys = liveTimingKeys(runId);
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
}
return keys;
}
if (ACP_TIMING_EVENTS.has(event)) {
const keys = liveTimingKeys(runId);
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
}
return keys;
}
if (event === "watchdog.timeout") {
return stageId ? [queryKeys.runs.stageEvents(runId, stageId)] : [];
}

View file

@ -3,6 +3,7 @@ import type { PaginatedRunStageList, StageHandler, StageState } from "@qltysh/fa
import type { Stage } from "../components/stage-sidebar";
import { aggregateGraphNodeStatus, formatStageLabel, mapRunStagesToSidebarStages } from "./stage-sidebar";
import { makeBilledTokenCounts } from "./test-fixtures";
function makeStage(nodeId: string, visit: number, status: StageState): Stage {
return {
@ -17,6 +18,7 @@ function makeStage(nodeId: string, visit: number, status: StageState): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: makeBilledTokenCounts(),
};
}
@ -38,6 +40,15 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
},
billing: makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
total_tokens: 43_690,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_usd_micros: 720_000,
}),
},
{
id: "apply-changes@2",
@ -46,6 +57,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "apply",
visit: 2,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -64,6 +76,10 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
});
// Each visit keeps its own tokens and cost, so the stage popover never
// shows a sibling visit's usage.
expect(result[0].billing.total_usd_micros).toBe(720_000);
expect(result[1].billing.total_usd_micros).toBeUndefined();
expect(formatStageLabel(result[0])).toBe("Apply Changes");
expect(result[1].id).toBe("apply-changes@2");
@ -83,6 +99,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "start",
visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "verify@1",
@ -91,6 +108,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "exit@1",
@ -99,6 +117,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "exit",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -118,6 +137,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -137,6 +157,7 @@ describe("mapRunStagesToSidebarStages", () => {
node_id: "work",
visit: 1,
graph_visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "work@2",
@ -147,6 +168,7 @@ describe("mapRunStagesToSidebarStages", () => {
visit: 2,
graph_visit: 1,
resumed_from_stage_id: "work@1",
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -172,6 +194,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -192,6 +215,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "pending",
node_id: "approval",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },

View file

@ -1,5 +1,6 @@
import { StageState } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
PaginatedRunStageList,
StageHandler,
StageModelUsage,
@ -27,6 +28,11 @@ export interface Stage {
resumedFromStageId: string | null;
startedAt: string | null;
providerUsed: StageModelUsage | null;
/**
* Tokens and cost for this visit alone, priced the same way the Billing tab
* prices its per-node rows. All-zero counts mean the stage called no model.
*/
billing: BilledTokenCounts;
}
export const ACTIVE_STAGE_STATES: ReadonlySet<StageState> = new Set([
@ -102,6 +108,7 @@ export function mapRunStagesToSidebarStages(
: "--",
startedAt: stage.started_at ?? null,
providerUsed: stage.provider_used ?? null,
billing: stage.billing,
});
}
return stages;

View file

@ -1,4 +1,4 @@
import type { Principal } from "@qltysh/fabro-api-client";
import type { BilledTokenCounts, Principal } from "@qltysh/fabro-api-client";
export const TEST_PRINCIPAL: Principal = {
kind: "user",
@ -6,3 +6,17 @@ export const TEST_PRINCIPAL: Principal = {
login: "test",
auth_method: "dev_token",
};
export function makeBilledTokenCounts(
overrides: Partial<BilledTokenCounts> = {},
): BilledTokenCounts {
return {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 0,
output_tokens: 0,
reasoning_tokens: 0,
total_tokens: 0,
...overrides,
};
}

View file

@ -1,14 +1,18 @@
import { useMemo } from "react";
import { useParams } from "react-router";
import { ArrowDownTrayIcon, PaperClipIcon } from "@heroicons/react/24/outline";
import { Disclosure, DisclosureButton, DisclosurePanel } from "@headlessui/react";
import { ArrowDownTrayIcon, ChevronRightIcon, PaperClipIcon } from "@heroicons/react/24/outline";
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import { EmptyState, ErrorState, LoadingState } from "../components/state";
import { StageSidebar } from "../components/stage-sidebar";
import { stageArtifactDownloadUrl } from "../lib/api-client";
import { formatBytes } from "../lib/format";
import { plural } from "../lib/plural";
import { useRunArtifacts, useRunStages } from "../lib/queries";
import { formatStageLabel, mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
import { mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
import type { ArtifactFile, ArtifactVersion } from "./run-artifacts/group";
import { groupArtifactsByFile } from "./run-artifacts/group";
export const handle = { wide: true };
@ -25,7 +29,12 @@ export default function RunArtifacts() {
<div className="flex gap-6">
<StageSidebar stages={stages} runId={id!} activeLink="artifacts" />
<div className="min-w-0 flex-1">
<RunArtifactsBody runId={id!} artifactsQuery={artifactsQuery} stages={stages} />
<RunArtifactsBody
runId={id!}
artifactsQuery={artifactsQuery}
stagesQuery={stagesQuery}
stages={stages}
/>
</div>
</div>
);
@ -34,86 +43,40 @@ export default function RunArtifacts() {
function RunArtifactsBody({
runId,
artifactsQuery,
stagesQuery,
stages,
}: {
runId: string;
artifactsQuery: ReturnType<typeof useRunArtifacts>;
stagesQuery: ReturnType<typeof useRunStages>;
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
}) {
if (artifactsQuery.error) {
const error = artifactsQuery.error ?? stagesQuery.error;
if (error) {
return (
<ErrorState
title="Couldn't load artifacts"
description={errorMessage(artifactsQuery.error)}
onRetry={() => void artifactsQuery.mutate()}
description={errorMessage(error)}
onRetry={() => {
if (artifactsQuery.error) void artifactsQuery.mutate();
if (stagesQuery.error) void stagesQuery.mutate();
}}
/>
);
}
if (artifactsQuery.data === undefined) {
if (artifactsQuery.data === undefined || stagesQuery.data === undefined) {
return <LoadingState label="Loading artifacts…" />;
}
const entries = artifactsQuery.data?.data ?? [];
if (entries.length === 0) {
return (
<EmptyState
icon={PaperClipIcon}
title="No artifacts captured"
description="No stage in this run produced any artifacts."
/>
);
}
return <ArtifactList runId={runId} entries={entries} stages={stages} />;
return (
<ArtifactFiles
runId={runId}
entries={artifactsQuery.data?.data ?? []}
stages={stages}
/>
);
}
interface StageGroup {
key: string;
stageId: string;
retry: number;
label: string;
entries: RunArtifactEntry[];
totalBytes: number;
}
function groupArtifacts(
entries: readonly RunArtifactEntry[],
stages: ReturnType<typeof mapRunStagesToSidebarStages>,
): StageGroup[] {
const stageLabels = new Map<string, string>();
for (const stage of stages) {
stageLabels.set(stage.id, formatStageLabel(stage));
}
const groups = new Map<string, StageGroup>();
for (const entry of entries) {
const key = `${entry.stage_id}#${entry.retry}`;
const existing = groups.get(key);
if (existing) {
existing.entries.push(entry);
existing.totalBytes += entry.size;
} else {
groups.set(key, {
key,
stageId: entry.stage_id,
retry: entry.retry,
label: stageLabels.get(entry.stage_id) ?? entry.node_slug,
entries: [entry],
totalBytes: entry.size,
});
}
}
for (const group of groups.values()) {
group.entries.sort((a, b) => a.relative_path.localeCompare(b.relative_path));
}
const sortedGroups = Array.from(groups.values());
sortedGroups.sort((a, b) => {
const labelCmp = a.label.localeCompare(b.label);
return labelCmp !== 0 ? labelCmp : a.retry - b.retry;
});
return sortedGroups;
}
function ArtifactList({
function ArtifactFiles({
runId,
entries,
stages,
@ -122,97 +85,209 @@ function ArtifactList({
entries: readonly RunArtifactEntry[];
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
}) {
const groups = useMemo(() => groupArtifacts(entries, stages), [entries, stages]);
const totalBytes = useMemo(
() => entries.reduce((sum, entry) => sum + entry.size, 0),
[entries],
);
const files = useMemo(() => groupArtifactsByFile(entries, stages), [entries, stages]);
if (files.length === 0) {
return (
<EmptyState
icon={PaperClipIcon}
title="No artifacts captured"
description="No stage in this run produced any artifacts."
/>
);
}
return <ArtifactList runId={runId} files={files} />;
}
function ArtifactList({ runId, files }: { runId: string; files: readonly ArtifactFile[] }) {
const { captures, latestBytes, storedBytes } = useMemo(() => {
let captures = 0;
let latestBytes = 0;
let storedBytes = 0;
for (const file of files) {
captures += file.versions.length;
latestBytes += file.versions[0].size;
for (const version of file.versions) storedBytes += version.size;
}
return { captures, latestBytes, storedBytes };
}, [files]);
// Only mention versions once some file actually has more than one.
const versioned = captures > files.length;
return (
<div className="space-y-4">
<div className="flex items-baseline justify-between">
<div className="flex flex-wrap items-baseline justify-between gap-4">
<h2 className="text-sm font-medium text-fg">
{entries.length} {entries.length === 1 ? "artifact" : "artifacts"}
{files.length} {plural(files.length, "file", "files")}
{versioned && (
<span className="font-normal text-fg-muted">
{" "}
· {captures} {plural(captures, "version", "versions")}
</span>
)}
</h2>
<span className="text-xs tabular-nums text-fg-muted">
{formatBytes(totalBytes)} total
<span className="text-xs text-fg-muted tabular-nums">
{versioned
? `${formatBytes(latestBytes)} latest · ${formatBytes(storedBytes)} stored`
: `${formatBytes(latestBytes)} total`}
</span>
</div>
{groups.map((group) => (
<StageGroupCard key={group.key} runId={runId} group={group} />
))}
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
{files.map((file) => (
<ArtifactFileRow key={file.path} runId={runId} file={file} />
))}
</section>
</div>
);
}
function StageGroupCard({ runId, group }: { runId: string; group: StageGroup }) {
function ArtifactFileRow({ runId, file }: { runId: string; file: ArtifactFile }) {
const hasEarlier = file.versions.length > 1;
const latest = file.versions[0];
return (
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
<header className="flex items-baseline justify-between border-b border-line px-4 py-2.5">
<div className="flex items-baseline gap-2">
<h3 className="text-sm font-medium text-fg">{group.label}</h3>
{group.retry > 0 && (
<span className="rounded bg-overlay px-1.5 py-0.5 text-[11px] font-medium text-fg-3">
retry {group.retry}
<Disclosure as="div" className="border-t border-line first:border-t-0">
{({ open }) => (
<>
<div className="flex items-center gap-2 px-3 py-2.5 sm:gap-4 sm:px-4">
{hasEarlier ? (
<DisclosureButton className="group shrink-0 rounded-md p-1 text-fg-3 transition-colors hover:bg-overlay hover:text-fg-2 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500">
<span className="sr-only">
{open ? "Hide" : "Show"} earlier versions of {file.name}
</span>
<ChevronRightIcon
className="size-3.5 transition-transform group-data-open:rotate-90"
aria-hidden="true"
/>
</DisclosureButton>
) : (
<span className="size-5 shrink-0" aria-hidden="true" />
)}
<span className="min-w-0 flex-1" title={file.path}>
<span className="block truncate font-mono text-xs">
<span className="text-fg-muted">{file.dir}</span>
<span className="text-fg-2">{file.name}</span>
</span>
<span className="mt-0.5 block truncate text-[11px] text-fg-3 md:hidden">
<VersionLabel version={latest} />
</span>
</span>
{hasEarlier && (
<span className="hidden shrink-0 rounded-full bg-overlay-strong px-2 py-0.5 text-[11px] text-fg-3 lg:inline">
{file.versions.length}{" "}
{plural(file.versions.length, "version", "versions")}
</span>
)}
<span className="hidden max-w-48 shrink-0 truncate text-xs text-fg-3 md:inline">
<VersionLabel version={latest} />
</span>
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
{formatBytes(latest.size)}
</span>
<DownloadLink runId={runId} file={file} version={latest} />
</div>
{hasEarlier && (
<DisclosurePanel
as="ul"
className="border-t border-line bg-black/15 py-1"
>
<EarlierVersions runId={runId} file={file} />
</DisclosurePanel>
)}
</div>
<span className="text-xs tabular-nums text-fg-muted">
{group.entries.length} {group.entries.length === 1 ? "file" : "files"}
{" · "}
{formatBytes(group.totalBytes)}
</span>
</header>
<ul className="divide-y divide-line">
{group.entries.map((entry) => (
<ArtifactRow
key={`${group.key}#${entry.relative_path}`}
runId={runId}
entry={entry}
/>
))}
</ul>
</section>
</>
)}
</Disclosure>
);
}
function ArtifactRow({ runId, entry }: { runId: string; entry: RunArtifactEntry }) {
function EarlierVersions({ runId, file }: { runId: string; file: ArtifactFile }) {
return (
<>
{file.versions.map((version, index) =>
index === 0 ? null : (
<li
key={`${version.stageId}#${version.retry}`}
className="flex items-center gap-2 py-1.5 pr-3 pl-10 hover:bg-overlay sm:gap-4 sm:pr-4 sm:pl-14"
>
<span className="min-w-0 flex-1 truncate text-xs text-fg-3">
<VersionLabel version={version} />
</span>
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
{formatBytes(version.size)}
</span>
<SizeDelta delta={version.delta} />
<DownloadLink runId={runId} file={file} version={version} />
</li>
),
)}
</>
);
}
function VersionLabel({ version }: { version: ArtifactVersion }) {
const attempt = attemptLabel(version);
return (
<>
{version.stageLabel}
{attempt && <span className="ml-2 text-fg-muted">{attempt}</span>}
</>
);
}
function attemptLabel(version: ArtifactVersion): string | null {
return version.retry > 1 ? `attempt ${version.retry}` : null;
}
function SizeDelta({ delta }: { delta: number | null }) {
if (delta === null) {
return <span className="shrink-0 text-[11px] text-fg-muted tabular-nums">first</span>;
}
const tone = delta < 0 ? "text-amber" : "text-mint";
const sign = delta < 0 ? "−" : "+";
return (
<span className={`shrink-0 text-[11px] ${tone} tabular-nums`}>
{sign}
{formatBytes(Math.abs(delta))}
</span>
);
}
function DownloadLink({
runId,
file,
version,
}: {
runId: string;
file: ArtifactFile;
version: ArtifactVersion;
}) {
const href = stageArtifactDownloadUrl(
runId,
entry.stage_id,
entry.relative_path,
entry.retry,
version.stageId,
file.path,
version.retry,
);
const attempt = attemptLabel(version);
const source = attempt ? `${version.stageLabel}, ${attempt}` : version.stageLabel;
return (
<li className="flex items-center gap-4 px-4 py-2">
<span
className="flex-1 truncate font-mono text-xs text-fg-2"
title={entry.relative_path}
>
{entry.relative_path}
</span>
<span className="shrink-0 tabular-nums text-xs text-fg-muted">
{formatBytes(entry.size)}
</span>
<a
href={href}
download={basename(entry.relative_path)}
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
>
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
Download
</a>
</li>
<a
href={href}
download={file.name}
aria-label={`Download ${file.name} from ${source}`}
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
>
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
<span className="hidden sm:inline">Download</span>
</a>
);
}
function basename(path: string): string {
const idx = path.lastIndexOf("/");
return idx >= 0 ? path.slice(idx + 1) : path;
}
function errorMessage(error: unknown): string | undefined {
return error instanceof Error ? error.message : undefined;
}

View file

@ -0,0 +1,192 @@
import { describe, expect, test } from "bun:test";
import { StageHandler, StageState } from "@qltysh/fabro-api-client";
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import type { Stage } from "../../lib/stage-sidebar";
import { groupArtifactsByFile, splitArtifactPath } from "./group";
function stage(nodeId: string, startedAt: string | null, visit = 1): Stage {
return {
id: `${nodeId}@${visit}`,
name: nodeId,
handler: StageHandler.AGENT,
nodeId,
visit,
graphVisit: null,
resumedFromStageId: null,
status: StageState.SUCCEEDED,
duration: "1s",
startedAt,
providerUsed: null,
};
}
function artifact(
nodeSlug: string,
path: string,
size: number,
retry = 1,
visit = 1,
): RunArtifactEntry {
return {
stage_id: `${nodeSlug}@${visit}`,
node_slug: nodeSlug,
retry,
relative_path: path,
size,
};
}
/** Mirrors run 01KYJ8ZR0N: one report rewritten by four stages. */
const REPORT = ".ai/reports/2026-07-27-wrk-002-instance-lifecycle.md";
const STAGES: Stage[] = [
stage("start", "2026-07-27T17:12:08Z"),
stage("plan", "2026-07-27T17:21:44Z"),
stage("implement_plan", "2026-07-27T17:44:18Z"),
stage("simplify", "2026-07-27T18:43:11Z"),
stage("consolidate_reviews", "2026-07-27T19:44:26Z"),
stage("fix_review_findings", "2026-07-27T19:49:22Z"),
];
describe("splitArtifactPath", () => {
test("splits a nested path into directory prefix and filename", () => {
expect(splitArtifactPath(".ai/reports/run.md")).toEqual({
dir: ".ai/reports/",
name: "run.md",
});
});
test("leaves a root-level path without a directory", () => {
expect(splitArtifactPath("README.md")).toEqual({ dir: "", name: "README.md" });
});
});
describe("groupArtifactsByFile", () => {
test("collapses repeated captures of one path into a single file", () => {
const files = groupArtifactsByFile(
[
artifact("consolidate_reviews", REPORT, 14323),
artifact("fix_review_findings", REPORT, 17483),
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
],
STAGES,
);
expect(files).toHaveLength(1);
expect(files[0].path).toBe(REPORT);
expect(files[0].dir).toBe(".ai/reports/");
expect(files[0].name).toBe("2026-07-27-wrk-002-instance-lifecycle.md");
expect(files[0].versions).toHaveLength(4);
});
test("orders versions newest first using the API stage order", () => {
const stages = [
stage("implement_plan", "2026-07-27T20:00:00Z"),
stage("simplify", "2026-07-27T18:43:11Z"),
];
const files = groupArtifactsByFile(
[
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
],
stages,
);
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
"simplify",
"implement_plan",
]);
expect(files[0].versions[0].size).toBe(13162);
});
test.each([
["equal", "2026-07-27T18:43:11Z", "2026-07-27T18:43:11Z"],
["missing", null, null],
])("preserves API order when stage timestamps are %s", (_case, firstAt, secondAt) => {
const stages = [
stage("implement_plan", firstAt),
stage("simplify", secondAt),
];
const files = groupArtifactsByFile(
[
artifact("simplify", REPORT, 13162),
artifact("implement_plan", REPORT, 8422),
],
stages,
);
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
"simplify",
"implement_plan",
]);
});
test("reports the byte change each capture introduced, oldest capture first", () => {
const files = groupArtifactsByFile(
[
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
artifact("consolidate_reviews", REPORT, 14323),
artifact("fix_review_findings", REPORT, 17483),
],
STAGES,
);
// versions are newest-first, so deltas read 17483-14323, 14323-13162, ...
expect(files[0].versions.map((v) => v.delta)).toEqual([3160, 1161, 4740, null]);
});
test("drops captures from graph control nodes", () => {
const files = groupArtifactsByFile(
[
artifact("start", ".ai/reports/pre-existing.md", 12402),
artifact("plan", ".ai/plans/plan.md", 21749),
],
STAGES,
);
expect(files.map((file) => file.path)).toEqual([".ai/plans/plan.md"]);
});
test("sorts files by their most recent capture", () => {
const files = groupArtifactsByFile(
[
artifact("plan", ".ai/plans/plan.md", 21749),
artifact("fix_review_findings", REPORT, 17483),
artifact("simplify", ".ai/reviews/bugs.xml", 5231),
],
STAGES,
);
expect(files.map((file) => file.path)).toEqual([
REPORT,
".ai/reviews/bugs.xml",
".ai/plans/plan.md",
]);
});
test("keeps retries of one stage as separate ordered versions", () => {
const files = groupArtifactsByFile(
[
artifact("simplify", REPORT, 13162, 2),
artifact("simplify", REPORT, 9000, 1),
],
STAGES,
);
expect(files[0].versions.map((v) => v.retry)).toEqual([2, 1]);
expect(files[0].versions[0].size).toBe(13162);
expect(files[0].versions.map((v) => v.delta)).toEqual([4162, null]);
});
test("returns no files when every capture came from a control node", () => {
const files = groupArtifactsByFile(
[artifact("start", ".ai/reports/pre-existing.md", 12402)],
STAGES,
);
expect(files).toEqual([]);
});
});

View file

@ -0,0 +1,104 @@
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import { isVisibleStage } from "../../data/runs";
import type { Stage } from "../../lib/stage-sidebar";
import { formatStageLabel } from "../../lib/stage-sidebar";
/** One capture of a file, written by a single stage attempt. */
export interface ArtifactVersion {
stageId: string;
stageLabel: string;
retry: number;
size: number;
/** Byte change this capture introduced; null for the first capture. */
delta: number | null;
}
/** One artifact path together with its capture history, newest first. */
export interface ArtifactFile {
path: string;
/** Directory prefix including the trailing slash, or "" at the root. */
dir: string;
name: string;
versions: readonly [ArtifactVersion, ...ArtifactVersion[]];
}
export function splitArtifactPath(path: string): { dir: string; name: string } {
const idx = path.lastIndexOf("/");
return idx >= 0
? { dir: path.slice(0, idx + 1), name: path.slice(idx + 1) }
: { dir: "", name: path };
}
interface StageInfo {
label: string;
order: number;
}
/** Stage display data keyed by ID, preserving the API's event order. */
function stageInfoById(stages: readonly Stage[]): Map<string, StageInfo> {
const info = new Map<string, StageInfo>();
stages.forEach((stage, order) => {
info.set(stage.id, { label: formatStageLabel(stage), order });
});
return info;
}
/**
* Collapse raw `(stage, retry, path)` capture keys into one entry per file,
* carrying the ordered history of every capture of that path.
*
* Captures from graph control nodes (`start`, `exit`) are dropped: those nodes
* run no work, so anything they match is a pre-existing workspace file rather
* than something the run produced.
*/
export function groupArtifactsByFile(
entries: readonly RunArtifactEntry[],
stages: readonly Stage[],
): ArtifactFile[] {
const stageInfo = stageInfoById(stages);
const byPath = new Map<string, [ArtifactVersion, ...ArtifactVersion[]]>();
for (const entry of entries) {
if (!isVisibleStage(entry.node_slug)) continue;
const info = stageInfo.get(entry.stage_id);
const version: ArtifactVersion = {
stageId: entry.stage_id,
stageLabel: info?.label ?? entry.node_slug,
retry: entry.retry,
size: entry.size,
delta: null,
};
const bucket = byPath.get(entry.relative_path);
if (bucket) bucket.push(version);
else byPath.set(entry.relative_path, [version]);
}
const files: Array<{ file: ArtifactFile; order: number }> = [];
for (const [path, versions] of byPath) {
// Oldest first, so each version's delta is the change that capture introduced.
versions.sort(
(a, b) =>
(stageInfo.get(a.stageId)?.order ?? -1) -
(stageInfo.get(b.stageId)?.order ?? -1) ||
a.retry - b.retry ||
a.stageId.localeCompare(b.stageId),
);
versions.forEach((version, index) => {
version.delta = index === 0 ? null : version.size - versions[index - 1].size;
});
versions.reverse();
const latest = versions[0];
const { dir, name } = splitArtifactPath(path);
files.push({
file: { path, dir, name, versions },
order: stageInfo.get(latest.stageId)?.order ?? -1,
});
}
// Most recently written file first — the page answers "what just happened?".
files.sort((a, b) => b.order - a.order || a.file.path.localeCompare(b.file.path));
return files.map((entry) => entry.file);
}

View file

@ -2,11 +2,12 @@ import { afterEach, describe, expect, mock, test } from "bun:test";
import TestRenderer from "react-test-renderer";
import type {
BilledTokenCounts,
RunBilling,
StageTiming,
} from "@qltysh/fabro-api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
function stageTiming(wall_time_ms = 0, inference_time_ms = 0, tool_time_ms = 0): StageTiming {
return {
wall_time_ms,
@ -24,25 +25,12 @@ mock.module("../lib/queries", () => ({
const { default: RunBillingRoute } = await import("./run-billing");
function zeroBilling(overrides: Partial<BilledTokenCounts> = {}): BilledTokenCounts {
return {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 0,
output_tokens: 0,
reasoning_tokens: 0,
total_tokens: 0,
total_usd_micros: null,
...overrides,
};
}
function billing(overrides: Partial<RunBilling> = {}): RunBilling {
return {
stages: [],
totals: {
timing: stageTiming(),
...zeroBilling(),
...makeBilledTokenCounts(),
},
by_model: [],
...overrides,
@ -86,21 +74,21 @@ describe("RunBilling", () => {
{
stage: { id: "start", name: "start" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(),
state: "succeeded",
},
{
stage: { id: "command", name: "command" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(61000),
state: "succeeded",
},
],
totals: {
timing: stageTiming(61000),
...zeroBilling(),
...makeBilledTokenCounts(),
},
}),
);
@ -121,7 +109,7 @@ describe("RunBilling", () => {
{
stage: { id: "start", name: "start" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(),
state: "succeeded",
},
@ -131,7 +119,7 @@ describe("RunBilling", () => {
provider: "anthropic",
model_id: "claude-sonnet-4-5",
},
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -143,7 +131,7 @@ describe("RunBilling", () => {
],
totals: {
timing: stageTiming(42000),
...zeroBilling({
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -157,7 +145,7 @@ describe("RunBilling", () => {
model_id: "claude-sonnet-4-5",
},
stages: 1,
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -204,7 +192,7 @@ describe("RunBilling", () => {
model_id: "claude-opus-4-6",
speed: "fast",
},
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -217,7 +205,7 @@ describe("RunBilling", () => {
],
totals: {
timing: stageTiming(),
...zeroBilling({
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -232,7 +220,7 @@ describe("RunBilling", () => {
speed: "fast",
},
stages: 1,
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -268,4 +256,4 @@ describe("RunBilling", () => {
Date.now = originalNow;
}
});
});
});

View file

@ -2,6 +2,11 @@ import { Fragment, useMemo } from "react";
import { EmptyState } from "../components/state";
import { Tooltip } from "../components/ui";
import {
billableOutputTokens,
billingTokenBuckets,
hasBillingUsage,
} from "../lib/billing";
import {
formatDurationMs,
formatTokenCount,
@ -11,6 +16,7 @@ import { useRunBilling } from "../lib/queries";
import { IN_FLIGHT_STAGE_STATES } from "../lib/stage-sidebar";
import { useTickingNow } from "../lib/time";
import type {
BilledTokenCounts,
BillingModelRef,
RunBilling,
RunBillingStage,
@ -39,23 +45,15 @@ function isInFlight(stage: RunBillingStage): boolean {
function isVisibleRow(row: MappedStageRow): boolean {
if (row.inFlight) return true;
return (
(row.inputTokens ?? 0) > 0 ||
(row.outputTokens ?? 0) > 0 ||
(row.totalUsdMicros ?? 0) > 0
);
return row.billing != null && hasBillingUsage(row.billing);
}
interface MappedStageRow {
stage: string;
model: string | null;
inputTokens: number | null;
outputTokens: number | null;
cacheReadTokens: number | null;
cacheWriteTokens: number | null;
wallTimeMs: number;
totalUsdMicros: number | null | undefined;
inFlight: boolean;
stage: string;
model: string | null;
billing: BilledTokenCounts | null;
wallTimeMs: number;
inFlight: boolean;
}
function liveWallTimeMs(stage: RunBillingStage, now: number): number {
@ -73,49 +71,28 @@ export const handle = { wide: true };
function mapStageRow(stage: RunBillingStage, wallTimeMs: number): MappedStageRow {
const hasModel = stage.model != null;
return {
stage: stage.stage.name,
model: formatModelRef(stage.model),
inputTokens: hasModel ? stage.billing.input_tokens : null,
outputTokens: hasModel
? stage.billing.output_tokens + stage.billing.reasoning_tokens
: null,
cacheReadTokens: hasModel ? stage.billing.cache_read_tokens : null,
cacheWriteTokens: hasModel ? stage.billing.cache_write_tokens : null,
stage: stage.stage.name,
model: formatModelRef(stage.model),
billing: hasModel ? stage.billing : null,
wallTimeMs,
totalUsdMicros: stage.billing.total_usd_micros,
inFlight: isInFlight(stage),
inFlight: isInFlight(stage),
};
}
/** Hover breakdown of the disjoint token buckets behind an `in / out` count. */
function TokenBreakdown({
cacheReadTokens,
cacheWriteTokens,
inputTokens,
outputTokens,
}: {
cacheReadTokens: number;
cacheWriteTokens: number;
inputTokens: number;
outputTokens: number;
}) {
const rows = [
{ label: "Cache read", value: cacheReadTokens },
{ label: "Cache creation", value: cacheWriteTokens },
{ label: "Uncached", value: inputTokens },
{ label: "Output", value: outputTokens },
];
function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
const buckets = billingTokenBuckets(billing);
return (
<div className="min-w-44 py-0.5">
<div className="mb-1.5 border-b border-line pb-1 font-medium text-fg-2">
<div className="border-line text-fg-2 mb-1.5 border-b pb-1 font-medium">
Tokens in / out
</div>
<dl className="grid grid-cols-[1fr_auto] gap-x-6 gap-y-1">
{rows.map((row) => (
<Fragment key={row.label}>
<dt className="text-fg-3">{row.label}</dt>
<dd className="text-right font-mono tabular-nums text-fg">
{formatTokens(row.value)}
{buckets.map((bucket) => (
<Fragment key={bucket.label}>
<dt className="text-fg-3">{bucket.label}</dt>
<dd className="text-fg text-right font-mono tabular-nums">
{formatTokens(bucket.value)}
</dd>
</Fragment>
))}
@ -128,42 +105,16 @@ function TokenBreakdown({
* Renders an `input / output` token count. When the row has model usage,
* hovering the count reveals the cache breakdown.
*/
function TokensCell({
inputTokens,
outputTokens,
cacheReadTokens,
cacheWriteTokens,
}: {
inputTokens: number | null;
outputTokens: number | null;
cacheReadTokens: number | null;
cacheWriteTokens: number | null;
}) {
function TokensCell({ billing }: { billing: BilledTokenCounts | null }) {
const display = (
<>
{formatTokens(inputTokens)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(outputTokens)}
{formatTokens(billing?.input_tokens)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(billing ? billableOutputTokens(billing) : null)}
</>
);
if (
inputTokens == null ||
outputTokens == null ||
cacheReadTokens == null ||
cacheWriteTokens == null
) {
return display;
}
if (!billing) return display;
return (
<Tooltip
label={
<TokenBreakdown
cacheReadTokens={cacheReadTokens}
cacheWriteTokens={cacheWriteTokens}
inputTokens={inputTokens}
outputTokens={outputTokens}
/>
}
>
<Tooltip label={<TokenBreakdown billing={billing} />}>
<span>{display}</span>
</Tooltip>
);
@ -189,15 +140,14 @@ export default function RunBilling({ params }: { params: { id: string } }) {
if (!billing) return [];
return billing.by_model
.map((entry) => ({
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
inputTokens: entry.billing.input_tokens,
outputTokens: entry.billing.output_tokens + entry.billing.reasoning_tokens,
cacheReadTokens: entry.billing.cache_read_tokens,
cacheWriteTokens: entry.billing.cache_write_tokens,
totalUsdMicros: entry.billing.total_usd_micros,
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
billing: entry.billing,
}))
.sort((a, b) => (b.totalUsdMicros ?? -1) - (a.totalUsdMicros ?? -1));
.sort(
(a, b) =>
(b.billing.total_usd_micros ?? -1) - (a.billing.total_usd_micros ?? -1),
);
}, [billing]);
// Re-derive only the in-flight rows on each tick; everything else stays put.
@ -218,16 +168,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
: (billing?.totals.timing.wall_time_ms ?? 0);
const hasLlmStages = (billing?.by_model.length ?? 0) > 0;
const totalInput = hasLlmStages ? (billing?.totals.input_tokens ?? null) : null;
const totalOutput = hasLlmStages && billing
? billing.totals.output_tokens + billing.totals.reasoning_tokens
: null;
const totalCacheRead = hasLlmStages
? (billing?.totals.cache_read_tokens ?? null)
: null;
const totalCacheWrite = hasLlmStages
? (billing?.totals.cache_write_tokens ?? null)
: null;
const totalBilling = hasLlmStages && billing ? billing.totals : null;
const totalUsdMicros = billing?.totals.total_usd_micros;
const modelStageCount = modelBreakdown.reduce((sum, row) => sum + row.stages, 0);
const visibleRows = rows.filter(isVisibleRow);
@ -268,18 +209,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.model ?? EMPTY_VALUE}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell
inputTokens={row.inputTokens}
outputTokens={row.outputTokens}
cacheReadTokens={row.cacheReadTokens}
cacheWriteTokens={row.cacheWriteTokens}
/>
<TokensCell billing={row.billing} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatDurationMs(row.wallTimeMs)}
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.totalUsdMicros)}
{formatUsdMicrosOrDash(row.billing?.total_usd_micros)}
</td>
</tr>
))}
@ -289,12 +225,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
<td className="px-4 py-3 font-medium text-fg">Total</td>
<td className="px-4 py-3 text-xs text-fg-muted">All models</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell
inputTokens={totalInput}
outputTokens={totalOutput}
cacheReadTokens={totalCacheRead}
cacheWriteTokens={totalCacheWrite}
/>
<TokensCell billing={totalBilling} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatDurationMs(totalWallTimeMs)}
@ -328,15 +259,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.stages}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell
inputTokens={row.inputTokens}
outputTokens={row.outputTokens}
cacheReadTokens={row.cacheReadTokens}
cacheWriteTokens={row.cacheWriteTokens}
/>
<TokensCell billing={row.billing} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.totalUsdMicros)}
{formatUsdMicrosOrDash(row.billing.total_usd_micros)}
</td>
</tr>
))}
@ -348,12 +274,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{modelStageCount}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell
inputTokens={totalInput}
outputTokens={totalOutput}
cacheReadTokens={totalCacheRead}
cacheWriteTokens={totalCacheWrite}
/>
<TokensCell billing={totalBilling} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatUsdMicrosOrDash(totalUsdMicros)}

View file

@ -19,6 +19,7 @@ import {
} from "../components/ui";
import { mutateRunListCaches } from "../lib/board-cache";
import { useDemoMode } from "../lib/demo-mode";
import { useTickingNow } from "../lib/time";
import { useSWRConfig } from "swr";
import {
useArchiveRun,
@ -56,10 +57,7 @@ import {
lifecycleActionVisibility,
updateLifecycleToastState,
} from "./run-detail/lifecycle-toasts";
import {
buildRunDetailRun,
useTickingNow,
} from "./run-detail/model";
import { buildRunDetailRun } from "./run-detail/model";
import {
buildRunDetailTabs,
childRouteLayoutFlags,
@ -69,6 +67,8 @@ import {
export const handle = { hideHeader: true };
const RUN_TIMING_REFRESH_INTERVAL_MS = 30_000;
type LifecycleTrigger = () => Promise<LifecycleMutationResult | undefined>;
export interface DockMeasurement {
@ -95,7 +95,7 @@ export function meta({ data }: any) {
export default function RunDetail({ params }: { params: { id: string } }) {
const demoMode = useDemoMode();
const runQuery = useRun(params.id);
const runQuery = useRun(params.id, RUN_TIMING_REFRESH_INTERVAL_MS);
const runStateQuery = useRunState(params.id);
const summary = runQuery.data;
const run = summary ? buildRunDetailRun(summary) : null;
@ -135,7 +135,10 @@ export default function RunDetail({ params }: { params: { id: string } }) {
childrenCount,
});
const steerBarRef = useRef<SteerBarHandle | null>(null);
const now = useTickingNow(30_000);
const now = useTickingNow(
summary != null && summary.timestamps.completed_at == null,
RUN_TIMING_REFRESH_INTERVAL_MS,
);
const { fullHeight, hideSteerBar } = childRouteLayoutFlags(matches);
useRunEvents(params.id);

View file

@ -327,6 +327,7 @@ function DurationPopover({
}) {
const endMs = completedAt != null ? Date.parse(completedAt) : now;
const sinceCreatedMs = Math.max(0, endMs - Date.parse(createdAt));
const isRunning = completedAt == null;
return (
<>
<PopoverHeader>Duration</PopoverHeader>
@ -336,8 +337,14 @@ function DurationPopover({
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(sinceCreatedMs)}</dd>
</div>
<div>
<dt className="text-fg-3">Active (inference + tools)</dt>
<dt className="text-fg-3">
Active (inference + tools){isRunning ? " — estimated" : ""}
</dt>
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(timing.active_time_ms)}</dd>
<dd className="mt-0.5 text-fg-3">
{formatDurationMs(timing.inference_time_ms)} inference ·{" "}
{formatDurationMs(timing.tool_time_ms)} tools
</dd>
</div>
</dl>
</>

View file

@ -1,6 +1,3 @@
import { useState } from "react";
import { useInterval } from "../../hooks/effects";
import {
isRunStatus,
mapRunToRunItem,
@ -8,12 +5,6 @@ import {
type Run,
} from "../../data/runs";
export function useTickingNow(intervalMs: number): number {
const [now, setNow] = useState(() => Date.now());
useInterval(() => setNow(Date.now()), intervalMs);
return now;
}
export type RunDetailRun = ReturnType<typeof mapRunToRunItem> & {
statusLabel: string;
statusDot: string;

View file

@ -3,6 +3,7 @@ import { renderToStaticMarkup } from "react-dom/server";
import { StageState } from "@qltysh/fabro-api-client";
import type { Stage } from "../lib/stage-sidebar";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { StageChatView } from "./run-stages";
function stage(overrides: Partial<Stage> = {}): Stage {
@ -18,6 +19,7 @@ function stage(overrides: Partial<Stage> = {}): Stage {
resumedFromStageId: null,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -1,9 +1,14 @@
import { describe, expect, test } from "bun:test";
import { renderToStaticMarkup } from "react-dom/server";
import type { ReasoningOutput } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
ReasoningOutput,
StageModelUsage,
} from "@qltysh/fabro-api-client";
import { EventDetails } from "./run-stages";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { EventDetails, ModelUsagePopover } from "./run-stages";
const RUN_START = "2026-04-09T12:00:00Z";
@ -72,3 +77,77 @@ describe("EventDetails reasoning", () => {
expect(html).toContain(`${"x".repeat(280)}…`);
});
});
const PROVIDER_USED: StageModelUsage = {
mode: "agent",
provider: "moonshot",
model: "kimi-k3",
reasoning_effort: "max",
};
function popoverMarkup(counts: BilledTokenCounts): string {
return renderToStaticMarkup(
<ModelUsagePopover providerUsed={PROVIDER_USED} billing={counts} />,
);
}
describe("ModelUsagePopover billing", () => {
test("shows the visit's token buckets and cost next to the model", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_tokens: 43_690,
total_usd_micros: 720_000,
}),
);
expect(html).toContain("kimi-k3");
expect(html).toContain("Cache read");
expect(html).toContain("4.8k");
expect(html).toContain("Cache creation");
expect(html).toContain("1.5k");
expect(html).toContain("Uncached");
expect(html).toContain("28.6k");
// Output folds in reasoning tokens, matching the Billing tab.
expect(html).toContain("Output");
expect(html).toContain("8.8k");
expect(html).toContain("Cost");
expect(html).toContain("$0.72");
});
test("omits the token section for a stage that called no model", () => {
const html = popoverMarkup(makeBilledTokenCounts());
expect(html).toContain("kimi-k3");
expect(html).not.toContain("Tokens");
expect(html).not.toContain("Cost");
});
test("still shows tokens when nothing priced the stage", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 1_000,
output_tokens: 500,
total_tokens: 1_500,
}),
);
expect(html).toContain("Uncached");
expect(html).toContain("1.0k");
expect(html).not.toContain("Cost");
});
test("shows a provider-reported cost when token counts are unavailable", () => {
const html = popoverMarkup(
makeBilledTokenCounts({ total_usd_micros: 720_000 }),
);
expect(html).toContain("kimi-k3");
expect(html).toContain("Cost");
expect(html).toContain("$0.72");
});
});

View file

@ -67,7 +67,9 @@ import {
formatBytes,
formatDurationMs,
formatTokenCount,
formatUsdMicros,
} from "../lib/format";
import { billingTokenBuckets, hasBillingUsage } from "../lib/billing";
import { plural } from "../lib/plural";
import {
useRun,
@ -93,6 +95,7 @@ import {
type UnknownRecord,
} from "../lib/unknown";
import type {
BilledTokenCounts,
EventEnvelope,
ReasoningOutput,
StageHandler,
@ -866,10 +869,42 @@ export function formatStageModelUsageLabel(
return effort ? `${model}[${effort}]` : model;
}
function ModelUsagePopover({
const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
/** Tokens and cost for this stage visit alone. */
function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
if (!hasBillingUsage(billing)) return null;
const buckets = billingTokenBuckets(billing);
const cost = formatUsdMicros(billing.total_usd_micros);
return (
<div className="mt-3">
<PopoverHeader>Tokens</PopoverHeader>
<PopoverRows>
{buckets.map((bucket) => (
<PopoverRow key={bucket.label} label={bucket.label}>
<span className={POPOVER_NUMBER}>
{bucket.value === 0
? "0"
: formatTokenCount(bucket.value, { compactDecimal: true })}
</span>
</PopoverRow>
))}
{cost && (
<PopoverRow label="Cost">
<span className={POPOVER_NUMBER}>{cost}</span>
</PopoverRow>
)}
</PopoverRows>
</div>
);
}
export function ModelUsagePopover({
providerUsed,
billing,
}: {
providerUsed: StageModelUsage;
billing: BilledTokenCounts;
}) {
return (
<>
@ -892,6 +927,7 @@ function ModelUsagePopover({
<PopoverRow label="Speed">{providerUsed.speed}</PopoverRow>
)}
</PopoverRows>
<StageBillingRows billing={billing} />
</>
);
}
@ -1905,6 +1941,7 @@ function EventsToolbar({
filteredCount,
totalCount,
providerUsed,
billing,
events,
runId,
stageId,
@ -1924,6 +1961,7 @@ function EventsToolbar({
filteredCount: number;
totalCount: number;
providerUsed: StageModelUsage | null;
billing: BilledTokenCounts;
events: EventEnvelope[];
runId: string;
stageId: string;
@ -2004,7 +2042,9 @@ function EventsToolbar({
className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${
showFilters ? "" : "ml-auto"
}`}
content={<ModelUsagePopover providerUsed={providerUsed} />}
content={
<ModelUsagePopover providerUsed={providerUsed} billing={billing} />
}
>
<CpuChipIcon className="size-3.5" aria-hidden="true" />
<span className="font-mono">{modelUsageLabel}</span>
@ -2359,6 +2399,7 @@ function RunStageActivityStage({
effectiveTab === "primary" ? turns.length : debugEvents.length
}
providerUsed={selectedStage.providerUsed}
billing={selectedStage.billing}
events={stageEventsQuery.data ?? []}
runId={runId}
stageId={selectedStageId}

View file

@ -26,6 +26,7 @@ import { ciConfig, columnForRun, columnStatusDisplay, columnStatuses, deriveCiSt
import type { CiStatus, CheckRun, CheckStatus, RunItem } from "../data/runs";
import { EmptyState } from "../components/state";
import { PullRequestChip } from "../components/pull-request-chip";
import { SizeChip } from "../components/size-chip";
import {
summarizeBatchLifecycleAction,
} from "../components/runs-list/batch-lifecycle";
@ -345,7 +346,7 @@ function PrCard({
// All inline footer metadata on PrCard belongs in this one row. Adding a new
// piece as a sibling `<div>` below the card body recreates a recurring bug
// where stats stack onto separate lines instead of sitting next to elapsed/actions.
// where stats stack onto separate lines instead of sitting next to size/actions.
function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
const hasActions = actions != null && actions.length > 0;
const hasStats =
@ -354,7 +355,7 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
(pr.additions != null && pr.additions !== 0) ||
(pr.deletions != null && pr.deletions !== 0);
if (!hasStats && !hasActions && pr.elapsed == null) return null;
if (!hasStats && !hasActions && pr.size == null) return null;
return (
<div className="mt-3 flex items-center gap-3 font-mono text-xs">
@ -416,9 +417,9 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
))}
</div>
)}
{pr.elapsed != null && (
<span className={`text-fg-muted ${hasActions ? "" : "ml-auto"}`}>
{pr.elapsed}
{pr.size != null && (
<span className={hasActions ? "inline-flex" : "ml-auto inline-flex"}>
<SizeChip size={pr.size} totalUsdMicros={pr.totalUsdMicros} />
</span>
)}
</div>

View file

@ -155,9 +155,16 @@ git diff --check
- Inference time is Fabro-observed LLM request/stream elapsed time, not
provider-reported model-only compute time.
- LLM retry backoff, queueing outside a request/stream, human waits, steering
waits, and scheduler gaps are wall time but not active time.
- Active timing is finalized-event based in v1; live active-time ticking can be
added later if it becomes necessary.
- Queueing outside a request/stream, human waits, steering waits, and scheduler
gaps are wall time but not active time. Retry delay inside an open LLM request
bracket follows the executor stopwatch and counts as inference time.
- ~~Active timing is finalized-event based in v1; live active-time ticking can
be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became
necessary: a run parked in one long agent stage reported ~12% of its wall time
as active, because in-flight stages contributed nothing. Stage projections now
accumulate inference and tool brackets from the event log and expose
`StageProjection::live_timing(now)`, the active-time twin of
`live_wall_time_ms`. Finalized values remain authoritative and still replace
the live estimate at terminal events. Implemented in PR #647.
- No compatibility layer is required for existing API clients or stored run
event data.

View file

@ -10673,7 +10673,38 @@ components:
- type: "null"
description: |
Per-attempt timing breakdown for the latest terminal attempt:
wall time plus the active inference/tool breakdown.
wall time plus the active inference/tool breakdown. Null while the
stage is still in flight; the live estimate is derived from
`live_inference_ms`, `live_tool_ms`, and any open bracket.
live_inference_ms:
type: integer
format: uint64
minimum: 0
default: 0
description: |
Inference time accumulated from closed brackets during the current
attempt. Live estimate only — the authoritative value arrives with
the terminal event and lands in `timing`. Excludes the currently
open bracket, whose span is measured from `inference.started_at`.
example: 78230
live_tool_ms:
type: integer
format: uint64
minimum: 0
default: 0
description: |
Tool time accumulated from closed tool batches during the current
attempt. A batch spans the first dispatched call through the
completion that drains the last outstanding one, so tools running
concurrently within a turn are counted once.
example: 7588
tool_batch:
oneOf:
- $ref: "#/components/schemas/StageToolBatchProjection"
- type: "null"
description: |
Open tool batch: when the batch started and which calls have not
yet reported completion.
usage:
$ref: "#/components/schemas/BilledTokenCounts"
model:
@ -10727,6 +10758,13 @@ components:
Open inference bracket, if the event log contains one. Present means
a model request was dispatched and no closing event has been seen —
not that the model is computing right now.
acp_started_at:
type: ["string", "null"]
format: date-time
description: >
Start of an external ACP agent process, if one is running. ACP
agents do not expose Fabro's internal LLM brackets, so the process
lifetime supplies their live inference estimate.
agent_control:
$ref: "#/components/schemas/AgentControlState"
description: Whether the agent is executing normally or waiting for steering after an interrupt.
@ -10734,6 +10772,37 @@ components:
$ref: "#/components/schemas/StageState"
description: Lifecycle state of the stage projection.
StageToolBatchProjection:
description: >
One open tool batch: tool calls dispatched together that have not all
reported completion. `open_call_ids` is a set rather than a count so a
duplicated completion in a replayed log cannot drain the batch early.
type: object
required:
- session_id
- started_at
- open_call_ids
properties:
session_id:
type: string
description: >
Root agent session that dispatched the batch. Transitions are
gated on it so delayed events from a replaced session cannot
mutate the current batch.
started_at:
type: string
format: date-time
description: >
When the batch opened — the first dispatched call observed while no
other calls were outstanding.
open_call_ids:
type: array
minItems: 1
uniqueItems: true
items:
type: string
description: Calls dispatched but not yet completed, by tool call id.
StageInferenceProjection:
description: >
One open inference bracket: a dispatched LLM request that has not yet
@ -12244,9 +12313,17 @@ components:
observed LLM request/stream elapsed time; `tool_time_ms` is tool or
command execution elapsed time; `active_time_ms` equals
`inference_time_ms + tool_time_ms`.
For a terminal stage these come from the worker's own stopwatch and are
authoritative. For a stage still in flight they are a live estimate
reconstructed from the event log, and `active_time_ms` is clamped to
`wall_time_ms`. The estimate is replaced by the authoritative
breakdown when the stage reaches a terminal event.
type: object
required:
- wall_time_ms
- inference_time_ms
- tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@ -12278,9 +12355,16 @@ components:
Timing rollup for an entire run. Active fields sum work across stage
visits, so `active_time_ms` can exceed `wall_time_ms` when parallel
branches run concurrently.
For a running run, stages still in flight contribute a live estimate
rather than nothing, so wall and active both advance continuously.
Unlike `StageTiming`, active is not clamped to wall here — concurrent
branches can legitimately sum past run wall time.
type: object
required:
- wall_time_ms
- inference_time_ms
- tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@ -12609,6 +12693,7 @@ components:
- status
- node_id
- visit
- billing
properties:
id:
$ref: "#/components/schemas/StageId"
@ -12668,6 +12753,15 @@ components:
format: date-time
description: Wall-clock time the latest attempt of this stage started, if known.
example: "2026-04-29T12:34:56Z"
billing:
$ref: "#/components/schemas/BilledTokenCounts"
description: >-
Token counts for this stage execution alone. `total_usd_micros` is
the provider-reported cost when there is one, otherwise the server
catalog's price for these tokens — the same pricing the
`/runs/{id}/billing` rows use. All-zero counts mean the stage made
no model calls. Unlike the billing rows, which sum every visit of a
node, this covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────

View file

@ -113,7 +113,7 @@ plan [label="Plan", prompt="Create an implementation plan."]
**Node identifiers** must start with a letter or underscore, followed by letters, digits, or underscores (e.g. `run_tests`, `gate_1`, `_private`).
Nodes referenced in edges are auto-created if not explicitly declared.
Every node used by an edge needs its own declaration. Validation fails when an edge names a node the workflow never declares, because that is nearly always a typo or a rename that missed an edge. The declaration can come before or after the edges that use it, and it can live in a subgraph.
### Edge declarations

View file

@ -61,11 +61,8 @@ pub(crate) async fn create_run(
None
};
let mut validation = manifest_validation::validate_manifest(
&RunLayer::default(),
&built.manifest,
ctx.catalog()?,
)?;
let mut validation =
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
let diagnostics = api_diagnostics_to_local(&validation.workflow.diagnostics);
if !quiet {

View file

@ -110,7 +110,6 @@ pub(crate) async fn execute(
run_id,
run_spec.source_directory.as_deref(),
&run_dir,
Arc::clone(&catalog),
)
} else {
None
@ -237,13 +236,12 @@ fn build_fabro_run_tool_services(
current_run_id: RunId,
source_directory: Option<&str>,
run_dir: &Path,
catalog: Arc<Catalog>,
) -> Option<FabroRunToolServices> {
if worker_token.trim().is_empty() {
return None;
}
let backend = ClientBackend::new(Arc::new(client))
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder { catalog }));
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder));
Some(FabroRunToolServices {
backend: Arc::new(backend),
current_run_id,
@ -252,9 +250,7 @@ fn build_fabro_run_tool_services(
})
}
struct WorkerRunManifestBuilder {
catalog: Arc<Catalog>,
}
struct WorkerRunManifestBuilder;
impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
fn build_run_manifest(
@ -263,12 +259,7 @@ impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> fabro_tool::ToolResult<RunManifest> {
run_tool_manifest::build_run_tool_manifest(
spec,
cwd,
user_settings_path,
Arc::clone(&self.catalog),
)
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}

View file

@ -23,11 +23,7 @@ pub(crate) fn run(
user_settings_path: Some(active_settings_path(None)),
..Default::default()
})?;
let response = manifest_validation::validate_manifest(
&RunLayer::default(),
&built.manifest,
base_ctx.catalog()?,
)?;
let response = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
let diagnostics = api_diagnostics_to_local(&response.workflow.diagnostics);
if base_ctx.json_output() {

View file

@ -49,54 +49,49 @@ where
pub(crate) fn print_diagnostics(diagnostics: &[Diagnostic], styles: &Styles, printer: Printer) {
for d in diagnostics {
let location = match (&d.node_id, &d.edge) {
(Some(node), _) => format!(" [node: {node}]"),
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
_ => String::new(),
};
let source_prefix = source_prefix(d);
match d.severity {
Severity::Error if source_prefix.is_empty() => fabro_util::printerr!(
printer,
"{}{location}: {} ({})",
styles.red.apply_to("error"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Error => fabro_util::printerr!(
printer,
"{}: {source_prefix}{}{location} ({})",
styles.red.apply_to("error"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!(
printer,
"{}{location}: {} ({})",
styles.yellow.apply_to("warning"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Warning => fabro_util::printerr!(
printer,
"{}: {source_prefix}{}{location} ({})",
styles.yellow.apply_to("warning"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Info => fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(if source_prefix.is_empty() {
format!("info{location}: {} ({})", d.message, d.rule)
} else {
format!("info: {source_prefix}{}{location} ({})", d.message, d.rule)
}),
),
print_diagnostic(d, styles, printer);
// The fix is the actionable half of a diagnostic, so it follows every
// severity rather than hiding behind --verbose. Rules that have nothing
// useful to suggest leave it unset.
if let Some(fix) = &d.fix {
fabro_util::printerr!(printer, " {} {fix}", styles.dim.apply_to("fix:"));
}
}
}
fn print_diagnostic(d: &Diagnostic, styles: &Styles, printer: Printer) {
let location = match (&d.node_id, &d.edge) {
(Some(node), _) => format!(" [node: {node}]"),
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
_ => String::new(),
};
let source_prefix = source_prefix(d);
let body = if source_prefix.is_empty() {
format!("{location}: {}", d.message)
} else {
format!(": {source_prefix}{}{location}", d.message)
};
match d.severity {
Severity::Error => fabro_util::printerr!(
printer,
"{}{body} ({})",
styles.red.apply_to("error"),
styles.dim.apply_to(&d.rule),
),
Severity::Warning => fabro_util::printerr!(
printer,
"{}{body} ({})",
styles.yellow.apply_to("warning"),
styles.dim.apply_to(&d.rule),
),
Severity::Info => fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(format!("info{body} ({})", d.rule)),
),
}
}
fn source_prefix(diagnostic: &Diagnostic) -> String {
match (
diagnostic.source_path.as_deref(),

View file

@ -101,6 +101,40 @@ fn create_uses_explicit_server_target_and_prints_remote_run_id() {
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
}
#[test]
fn create_defers_provider_validation_to_the_server() {
let context = test_context!();
let server = MockServer::start();
let run_id = unique_run_id();
let mock = server.mock(|when, then| {
when.method("POST")
.path("/api/v1/runs")
.body_includes(r#"provider=\"server-only\""#);
then.status(201)
.header("Content-Type", "application/json")
.body(run_status_response(run_id.as_str(), "submitted").to_string());
});
let output = context
.create_cmd()
.args([
"--server",
&format!("{}/api/v1", server.base_url()),
"--dry-run",
fixture("server-model.fabro").to_str().unwrap(),
])
.output()
.expect("command should execute");
assert!(
output.status.success(),
"local validation should not reject a server-owned provider\nstdout:\n{}\nstderr:\n{}",
String::from_utf8_lossy(&output.stdout),
String::from_utf8_lossy(&output.stderr)
);
mock.assert();
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
}
#[test]
fn create_uses_configured_server_target_without_server_flag() {
let context = test_context!();

View file

@ -95,7 +95,9 @@ fn graph_allow_invalid_renders_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
");
let svg = read_text(&output_path);
@ -119,7 +121,9 @@ fn graph_invalid_workflow_fails_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}

View file

@ -52,7 +52,9 @@ fn preflight_invalid_workflow_fails_with_validation_output() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}
@ -74,7 +76,9 @@ fn preflight_rejects_unbound_template_inputs() {
Goal: Demo
error: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
error: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
× Validation failed
");
}

View file

@ -69,6 +69,24 @@ fn simple_does_not_connect_to_configured_server() {
);
}
/// Offline validation has no model catalog, so a model or provider the server
/// owns must pass rather than be reported as unknown.
#[test]
fn server_owned_provider_is_not_rejected_by_offline_validation() {
let context = test_context!();
let mut cmd = context.validate();
cmd.arg(fixture("server-model.fabro"));
fabro_snapshot!(context.filters(), cmd, @"
success: true
exit_code: 0
----- stdout -----
----- stderr -----
Workflow: ServerModel (3 nodes, 2 edges)
Graph: [FIXTURES]/server-model.fabro
Validation: OK
");
}
#[test]
fn branching() {
let context = test_context!();
@ -82,6 +100,7 @@ fn branching() {
Workflow: Branch (6 nodes, 6 edges)
Graph: [FIXTURES]/branching.fabro
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
fix: Add retry_target or fallback_retry_target attribute
Validation: OK
");
}
@ -163,7 +182,9 @@ fn bare_fabro_with_unbound_inputs_validates_structurally_with_warning() {
Workflow: TemplatedUnbound (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound.fabro
warning: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
warning: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
Validation: OK
");
}
@ -186,6 +207,7 @@ fn bare_fabro_with_unbound_inputs_in_imported_prompt_validates_structurally_with
Workflow: TemplatedUnboundImported (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_imported/workflow.fabro
warning: [FIXTURES]/templated_unbound_imported/work.md:1:12: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
Validation: OK
");
}
@ -207,6 +229,7 @@ fn bare_fabro_with_unbound_inputs_in_template_partial_validates_structurally_wit
Workflow: TemplatedUnboundPartial (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_partial/workflow.fabro
warning: [FIXTURES]/templated_unbound_partial/test-include.partial.md:1:4: undefined template variable `inputs.hello` in node `test_imported_include` attribute `prompt` [node: test_imported_include] (template_undefined_variable)
fix: bind `inputs.hello` via `[run.inputs]` in workflow.toml, or pass `--input inputs.hello=<value>`
Validation: OK
");
}
@ -278,6 +301,26 @@ fn validate_reports_missing_template_dependency() {
");
}
/// A node named only by an edge is almost always a typo, so validation must
/// fail instead of quietly running it as a default agent stage.
#[test]
fn edge_only_node() {
let context = test_context!();
let mut cmd = context.validate();
cmd.arg(fixture("edge_only_node.fabro"));
fabro_snapshot!(context.filters(), cmd, @"
success: false
exit_code: 1
----- stdout -----
----- stderr -----
Workflow: EdgeOnlyNode (2 nodes, 2 edges)
Graph: [FIXTURES]/edge_only_node.fabro
error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists)
fix: Declare node 'misspelled_node' or correct the edge endpoint
× Validation failed
");
}
#[test]
fn invalid() {
let context = test_context!();
@ -291,7 +334,9 @@ fn invalid() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}

View file

@ -19,6 +19,7 @@ fn dry_run_branching() {
Goal: Implement and validate a feature
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
fix: Add retry_target or fallback_retry_target attribute
Run: [ULID]
Web UI: http://localhost:3000/runs/[ULID]
Sandbox: local (ready in [TIME])

View file

@ -1,11 +1,8 @@
use std::path::Path;
use std::sync::Arc;
use fabro_api::types;
use fabro_config::load_llm_catalog_settings;
use fabro_model::Catalog;
use fabro_server::run_tool_manifest;
use fabro_tool::{RunManifestBuilder, ToolError, ToolResult, ValidatedCreateRunSpec};
use fabro_tool::{RunManifestBuilder, ToolResult, ValidatedCreateRunSpec};
#[derive(Default)]
pub(crate) struct McpRunManifestBuilder;
@ -17,20 +14,6 @@ impl RunManifestBuilder for McpRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult<types::RunManifest> {
build_mcp_run_manifest(spec, cwd, user_settings_path)
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}
fn build_mcp_run_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult<types::RunManifest> {
let llm_catalog_settings = load_llm_catalog_settings(Some(user_settings_path))
.map_err(|err| ToolError::message(err.to_string()))?;
let catalog = Arc::new(
Catalog::from_builtin_with_overrides(&llm_catalog_settings)
.map_err(|err| ToolError::message(err.to_string()))?,
);
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, catalog)
}

View file

@ -1134,6 +1134,7 @@ mod runs {
id: stage_id.clone(),
name: name.to_owned(),
handler,
billing: BilledTokenCounts::default(),
status,
wall_time_ms,
node_id: stage_id.node_id().to_owned(),

View file

@ -1,41 +1,30 @@
use std::collections::HashMap;
use std::sync::Arc;
use anyhow::Result;
use fabro_api::types;
use fabro_config::{EnvironmentLayer, MergeMap, RunLayer};
use fabro_model::Catalog;
use fabro_config::RunLayer;
use fabro_workflow::pipeline::TEMPLATE_UNDEFINED_VARIABLE_RULE;
use crate::run_manifest;
/// Validate a manifest without a model catalog.
///
/// Every caller is a client — the CLI, an MCP server, a run worker — and a
/// client's catalog is its own, not the server's. Judging model and provider
/// availability here would reject workflows the server can run, so that is
/// left to the server on create.
pub fn validate_manifest(
manifest_run_defaults: &RunLayer,
manifest: &types::RunManifest,
catalog: Arc<Catalog>,
) -> Result<types::ValidateResponse> {
validate_manifest_with_environment_defaults(
manifest_run_defaults,
&fabro_environment::seeded_catalog_layer(),
manifest,
catalog,
)
}
pub fn validate_manifest_with_environment_defaults(
manifest_run_defaults: &RunLayer,
manifest_environment_defaults: &MergeMap<EnvironmentLayer>,
manifest: &types::RunManifest,
catalog: Arc<Catalog>,
) -> Result<types::ValidateResponse> {
let prepared = run_manifest::prepare_manifest_with_environment_defaults(
manifest_run_defaults,
manifest_environment_defaults,
&fabro_environment::seeded_catalog_layer(),
&HashMap::new(),
manifest,
)?;
let validated =
run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?;
let validated = run_manifest::validate_prepared_manifest_structural(&prepared)
.map_err(anyhow::Error::new)?;
Ok(run_manifest::validate_response(&prepared, &validated))
}

View file

@ -35,7 +35,8 @@ use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSecti
use fabro_validate::Severity;
use fabro_workflow::Error as WorkflowError;
use fabro_workflow::operations::{
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_ready_providers,
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_catalog,
validate_with_ready_providers,
};
use fabro_workflow::pipeline::Validated;
use fabro_workflow::run_materialization::materialize_run_with_ready_providers;
@ -192,12 +193,18 @@ pub(crate) fn validate_prepared_manifest(
validate_prepared_manifest_with_vars(prepared, catalog, HashMap::new())
}
pub(crate) fn validate_prepared_manifest_structural(
prepared: &PreparedManifest,
) -> Result<Validated, WorkflowError> {
validate(manifest_validate_input(prepared, HashMap::new()))
}
pub(crate) fn validate_prepared_manifest_with_vars(
prepared: &PreparedManifest,
catalog: Arc<Catalog>,
vars: HashMap<String, String>,
) -> Result<Validated, WorkflowError> {
validate(manifest_validate_input(prepared, catalog, vars))
validate_with_catalog(manifest_validate_input(prepared, vars), catalog)
}
pub(crate) fn validate_prepared_manifest_for_preflight(
@ -207,14 +214,14 @@ pub(crate) fn validate_prepared_manifest_for_preflight(
ready_providers: &[ProviderId],
) -> Result<Validated, WorkflowError> {
validate_with_ready_providers(
manifest_validate_input(prepared, catalog, vars),
manifest_validate_input(prepared, vars),
catalog,
ready_providers,
)
}
fn manifest_validate_input(
prepared: &PreparedManifest,
catalog: Arc<Catalog>,
vars: HashMap<String, String>,
) -> ValidateInput {
ValidateInput {
@ -223,7 +230,6 @@ fn manifest_validate_input(
vars,
cwd: prepared.cwd.clone(),
custom_transforms: Vec::new(),
catalog,
}
}

View file

@ -1,20 +1,22 @@
use std::path::{Path, PathBuf};
use std::sync::Arc;
use fabro_api::types;
use fabro_config::{CliLayer, RunGoalLayer, RunLayer};
use fabro_manifest::{ManifestBuildInput, RunOverrideInput};
use fabro_model::Catalog;
use fabro_tool::{ToolError, ToolResult, ValidatedCreateRunSpec};
use fabro_types::settings::interp::InterpString;
use crate::manifest_validation;
/// Build and validate a run manifest for the `fabro_run_create` tool.
///
/// Validation is structural. The caller is a client — an MCP server or a run
/// worker — whose catalog is its own, not the server's, so judging model and
/// provider availability here would reject workflows the server can run.
pub fn build_run_tool_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
catalog: Arc<Catalog>,
) -> ToolResult<types::RunManifest> {
let built = fabro_manifest::build_run_manifest(ManifestBuildInput {
workflow: PathBuf::from(&spec.workflow),
@ -30,7 +32,7 @@ pub fn build_run_tool_manifest(
.map_err(|err| ToolError::from_anyhow(&err))?;
let mut validation =
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest, catalog)
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)
.map_err(|err| ToolError::from_anyhow(&err))?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
if !validation.ok {
@ -103,6 +105,63 @@ mod tests {
use super::*;
fn create_run_spec(workflow: &str) -> ValidatedCreateRunSpec {
ValidatedCreateRunSpec::try_from(CreateRunSpec {
workflow: workflow.to_string(),
run_id: None,
parent_id: None,
cwd: None,
goal: None,
goal_file: None,
inputs: HashMap::new(),
labels: HashMap::new(),
model: None,
provider: None,
environment: None,
dry_run: None,
auto_approve: None,
preserve_sandbox: None,
start: None,
})
.expect("create spec should validate")
}
/// The tool runs on a client, whose catalog is not the server's, so a
/// server-owned model must reach the server rather than fail here.
#[expect(
clippy::disallowed_methods,
reason = "sync test writes one workflow fixture before building the manifest"
)]
#[test]
fn server_owned_provider_is_not_rejected_by_tool_manifest_validation() {
let dir = tempfile::tempdir().expect("temp dir should be created");
let workflow = dir.path().join("server-model.fabro");
std::fs::write(
&workflow,
r#"digraph ServerModel {
graph [goal="Use a server-owned model"]
start [shape=Mdiamond]
work [prompt="Do work", model="private-model", provider="server-only"]
exit [shape=Msquare]
start -> work -> exit
}"#,
)
.expect("workflow fixture should be written");
let manifest = build_run_tool_manifest(
&create_run_spec(&workflow.to_string_lossy()),
dir.path(),
&dir.path().join("settings.toml"),
)
.expect("tool validation should leave provider availability to the server");
let encoded = serde_json::to_string(&manifest).expect("manifest should serialize");
assert!(
encoded.contains("server-only"),
"the authored provider should survive into the manifest: {encoded}"
);
}
#[test]
fn manifest_args_preserve_input_provenance() {
let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec {

View file

@ -2,6 +2,7 @@ use std::collections::HashMap;
use std::sync::Arc;
use chrono::{DateTime, Utc};
use fabro_model::Catalog;
use fabro_types::{
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
};
@ -22,6 +23,7 @@ fn run_stage_from_projection(
stage_id: &StageId,
stage: &StageProjection,
graph: &Graph,
catalog: &Catalog,
now: DateTime<Utc>,
) -> RunStage {
let handler = stage.handler.unwrap_or_else(|| {
@ -36,6 +38,7 @@ fn run_stage_from_projection(
id: stage_id.clone(),
name: stage_id.node_id().to_owned(),
handler,
billing: stage.billed_usage(Some(catalog)).into_owned(),
status: stage.effective_state(),
wall_time_ms: stage.live_wall_time_ms(now),
node_id: stage_id.node_id().to_owned(),
@ -67,9 +70,10 @@ async fn list_run_stages(
let now = Utc::now();
let graph = projection.spec().graph();
let catalog = state.catalog();
let stages = projection
.iter_stages()
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, now))
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, &catalog, now))
.collect::<Vec<_>>();
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()
@ -175,7 +179,7 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
index
});
let row = &mut rows[index];
let stage_timing = billing_stage_timing(stage, now);
let stage_timing = stage.live_timing(now);
row.timing = row.timing.saturating_add(&stage_timing);
if stage_id.visit() >= row.latest_visit {
@ -188,20 +192,6 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
rows
}
/// Per-visit timing for a stage. For terminal visits, the stored breakdown is
/// used directly. For in-flight visits, fall back to the live wall-clock since
/// `started_at` (no active breakdown yet — that is only finalized at terminal
/// event time in v1).
fn billing_stage_timing(stage: &StageProjection, now: DateTime<Utc>) -> StageTiming {
if let Some(timing) = stage.timing {
return timing;
}
if let Some(live_wall) = stage.live_wall_time_ms(now) {
return StageTiming::wall_only(live_wall);
}
StageTiming::default()
}
fn stage_has_billing_row(stage: &StageProjection) -> bool {
stage.completion.is_some()
|| stage.timing.is_some()

View file

@ -11,7 +11,7 @@ use axum_extra::extract::Query as ExtraQuery;
use base64::Engine as _;
use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
use bytes::Bytes;
use chrono::Utc;
use chrono::{DateTime, Utc};
use fabro_api::types::{
BoardColumn, RunManifest, SubmitAnswerRequest, UpdateRunParentRequest, UpdateRunRequest,
};
@ -23,7 +23,7 @@ use fabro_store::{
};
use fabro_types::settings::ResolveError;
use fabro_types::{
AutomationRef, Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
AutomationRef, Principal, Run, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
RunStatusKind, StageContextWindow, StageContextWindowStaleness,
StageContextWindowUnavailableReason, StageHandler, StageModelUsage, StageProjection,
SystemActorKind, WorkflowSettings, parse_blob_ref,
@ -298,7 +298,7 @@ async fn validate_parent_link(
}
async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
match state.stores.run_summaries.get(run_id, Utc::now()).await {
match run_summary_at(state, run_id, Utc::now()).await {
Ok(Some(summary)) => (
StatusCode::OK,
Json(state.decorate_run_summary(summary).await),
@ -311,6 +311,28 @@ async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
}
}
/// Read the durable summary and overlay its timing from the live projection.
///
/// The SQLite read model stores active timing as of the most recent event.
/// An open inference or tool bracket keeps accruing between events, so detail
/// reads need the projection's current estimate while the run is non-terminal.
async fn run_summary_at(
state: &AppState,
run_id: &RunId,
now: DateTime<Utc>,
) -> fabro_store::Result<Option<Run>> {
let Some(mut summary) = state.stores.run_summaries.get(run_id, now).await? else {
return Ok(None);
};
if summary.timestamps.completed_at.is_none() {
let projection = state.stores.runs.get_cached_projection(run_id).await?;
if let Some(timing) = projection.and_then(|projection| projection.live_run_timing(now)) {
summary.timing = Some(timing);
}
}
Ok(Some(summary))
}
async fn list_runs(
_auth: RequiredRunManagementActor,
State(state): State<Arc<AppState>>,
@ -934,7 +956,7 @@ async fn get_run_status(
RequireRunManagementTarget(id, _actor): RequireRunManagementTarget,
State(state): State<Arc<AppState>>,
) -> Response {
match state.stores.run_summaries.get(&id, Utc::now()).await {
match run_summary_at(&state, &id, Utc::now()).await {
Ok(Some(run)) => {
(StatusCode::OK, Json(state.decorate_run_summary(run).await)).into_response()
}

View file

@ -5259,6 +5259,64 @@ fn test_billed_usage(
.unwrap()
}
async fn create_billed_retry_run(state: &Arc<AppState>, run_id: RunId) {
create_durable_run_with_events(state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
state,
run_id,
"verify",
1,
&workflow_event::Event::StageFailed {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(1200),
billing: Some(test_billed_usage("gpt-old", 100, 10)),
actor: None,
},
)
.await;
append_scoped_stage_event(
state,
run_id,
"verify",
2,
&workflow_event::Event::StageCompleted {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
timing: fabro_types::StageTiming::wall_only(800),
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing: Some(test_billed_usage("gpt-new", 200, 20)),
failure: None,
notes: None,
files_touched: Vec::new(),
context_updates: None,
jump_to_node: None,
context_values: None,
node_visits: None,
loop_failure_signatures: None,
restart_failure_signatures: None,
response: None,
attempt: 2,
max_attempts: 2,
},
)
.await;
}
#[tokio::test]
async fn list_run_stages_distinguishes_visits() {
let state = test_app_state_with_isolated_storage();
@ -5503,6 +5561,50 @@ async fn list_run_stages_exposes_execution_identity_for_resumed_stage() {
assert_eq!(second["resumed_from_stage_id"], "work@1");
}
#[tokio::test]
async fn run_billing_includes_live_stage_timing_in_rows_and_totals() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
workflow_run_started_event(run_id),
])
.await;
append_scoped_stage_event(
&state,
run_id,
"work",
1,
&stage_started_event("work", "command"),
)
.await;
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/billing")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let stages = body["stages"].as_array().unwrap();
assert_eq!(stages.len(), 1);
let row_timing = &stages[0]["timing"];
assert!(row_timing["active_time_ms"].as_u64().unwrap() > 0);
assert_eq!(row_timing["tool_time_ms"], row_timing["active_time_ms"]);
assert_eq!(&body["totals"]["timing"], row_timing);
}
/// `checkpoint.completed_nodes` records every visit, so a looped node appears
/// once per re-entry. Billing must dedup so a retried node renders as one row
/// and `runtime_secs` is summed across all visits exactly once.
@ -5654,65 +5756,9 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
let failed_usage = test_billed_usage("gpt-old", 100, 10);
create_billed_retry_run(&state, run_id).await;
let success_usage = test_billed_usage("gpt-new", 200, 20);
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
1,
&workflow_event::Event::StageFailed {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(1200),
billing: Some(failed_usage),
actor: None,
},
)
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
2,
&workflow_event::Event::StageCompleted {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
timing: fabro_types::StageTiming::wall_only(800),
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing: Some(success_usage.clone()),
failure: None,
notes: None,
files_touched: Vec::new(),
context_updates: None,
jump_to_node: None,
context_values: None,
node_visits: None,
loop_failure_signatures: None,
restart_failure_signatures: None,
response: None,
attempt: 2,
max_attempts: 2,
},
)
.await;
let mut latest_outcome: Outcome<Option<fabro_model::BilledModelUsage>> = Outcome::success();
latest_outcome.usage = Some(success_usage);
latest_outcome.timing = Some(fabro_types::StageTiming::wall_only(800));
@ -5746,7 +5792,6 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
.unwrap();
let response = app
.clone()
.oneshot(
Request::builder()
.method("GET")
@ -5791,6 +5836,76 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
assert_eq!(new_model["billing"]["input_tokens"], 200);
}
/// The stage popover reads `billing` off the stages list, so it must be scoped
/// to one visit — unlike the Billing tab's rows, which sum every visit of a
/// node.
#[tokio::test]
async fn list_run_stages_reports_billing_per_visit() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_billed_retry_run(&state, run_id).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let first = stage_entry(&body, "verify@1");
assert_eq!(first["billing"]["input_tokens"], 100);
assert_eq!(first["billing"]["output_tokens"], 10);
assert_eq!(first["billing"]["total_usd_micros"], 110);
let second = stage_entry(&body, "verify@2");
assert_eq!(second["billing"]["input_tokens"], 200);
assert_eq!(second["billing"]["output_tokens"], 20);
assert_eq!(second["billing"]["total_usd_micros"], 220);
}
#[tokio::test]
async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
let started = stage_started_event("script", "command");
append_scoped_stage_event(&state, run_id, "script", 1, &started).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let billing = &stage_entry(&body, "script@1")["billing"];
assert_eq!(billing["input_tokens"], 0);
assert_eq!(billing["output_tokens"], 0);
// No model ran, so there is nothing to price — not a $0.00 cost.
assert!(billing.get("total_usd_micros").is_none());
}
#[tokio::test]
async fn list_run_stages_shows_retrying_after_failed_event() {
let state = test_app_state_with_isolated_storage();
@ -7800,6 +7915,51 @@ async fn get_run_status_returns_status() {
assert!(body["labels"].is_object());
}
#[tokio::test]
async fn get_run_status_advances_live_active_timing_between_events() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
workflow_run_started_event(run_id),
])
.await;
append_scoped_stage_event(
&state,
run_id,
"work",
1,
&stage_started_event("work", "command"),
)
.await;
// The SQLite summary stores timing at the StageStarted event. A later
// detail read must overlay the in-flight command's active time from the
// projection even though no newer event has arrived.
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let timing = &body["timing"];
assert!(timing["active_time_ms"].as_u64().unwrap() > 0);
assert_eq!(timing["tool_time_ms"], timing["active_time_ms"]);
assert!(timing["wall_time_ms"].as_u64().unwrap() >= timing["active_time_ms"].as_u64().unwrap());
}
#[tokio::test]
async fn get_run_status_not_found() {
let app = test_app_with();

View file

@ -13,6 +13,7 @@ use axum::middleware::Next;
use axum::response::Response;
use axum::{Router, middleware};
use chrono::Duration as ChronoDuration;
use fabro_config::user::default_storage_dir;
use fabro_config::{RunLayer, ServerSettingsBuilder, Storage, envfile};
use fabro_db::DbPool;
use fabro_interview::Interviewer;
@ -253,9 +254,10 @@ impl TestAppStateBuilder {
self.try_build().expect("test app state should build")
}
pub fn try_build(self) -> anyhow::Result<Arc<AppState>> {
pub fn try_build(mut self) -> anyhow::Result<Arc<AppState>> {
let (store, artifact_store) = self.store_bundle.unwrap_or_else(test_store_bundle);
let vault_path = self.vault_path.unwrap_or_else(test_secret_store_path);
self.server_settings = redirect_default_storage_root(self.server_settings, &vault_path);
if !self.vault_entries.is_empty() {
let mut vault = Vault::load(vault_path.clone()).expect("test vault should load");
for (name, value) in &self.vault_entries {
@ -672,6 +674,27 @@ pub fn test_secret_store_path() -> PathBuf {
dir.join("secrets.json")
}
/// Keeps tests off the developer's real `~/.fabro/storage`.
///
/// Settings built for tests usually omit `[server.storage] root`, which
/// resolves to the production default. Handlers that walk that tree — `df`,
/// `system/resources`, `prune` — then read whatever runs and scratch
/// directories the machine happens to have, making tests slow and
/// machine-dependent, and letting run-creating tests write there.
///
/// Only settings still carrying the production default are redirected; a test
/// that chose its own root keeps it. The redirect goes through
/// [`ServerSettings::with_storage_override`] so the derived local object-store
/// roots move with it instead of pointing back at the real storage tree.
fn redirect_default_storage_root(settings: ServerSettings, vault_path: &Path) -> ServerSettings {
if Path::new(&settings.server.storage.root) != default_storage_dir() {
return settings;
}
let root = vault_path.with_file_name("storage");
std::fs::create_dir_all(&root).expect("test storage root should be creatable");
settings.with_storage_override(&root)
}
#[must_use]
pub fn test_auth_mode() -> AuthMode {
AuthMode::Enabled(ConfiguredAuth {

View file

@ -1,4 +1,4 @@
use std::collections::HashMap;
use std::collections::{HashMap, HashSet};
use std::time::Duration;
use crate::error::Error;
@ -57,29 +57,44 @@ fn derive_class_from_label(label: &str) -> String {
.collect()
}
fn collect_declared_node_ids(statements: &[Statement], node_ids: &mut HashSet<String>) {
for statement in statements {
match statement {
Statement::Node(node) => {
node_ids.insert(node.id.clone());
}
Statement::Subgraph(subgraph) => {
collect_declared_node_ids(&subgraph.statements, node_ids);
}
_ => {}
}
}
}
struct SemanticState {
graph: Graph,
node_defaults: HashMap<String, AttrValue>,
edge_defaults: HashMap<String, AttrValue>,
graph: Graph,
declared_node_ids: HashSet<String>,
node_defaults: HashMap<String, AttrValue>,
edge_defaults: HashMap<String, AttrValue>,
}
impl SemanticState {
fn new(name: String) -> Self {
fn new(name: String, declared_node_ids: HashSet<String>) -> Self {
Self {
graph: Graph::new(name),
graph: Graph::new(name),
declared_node_ids,
node_defaults: HashMap::new(),
edge_defaults: HashMap::new(),
}
}
fn ensure_node(&mut self, id: &str) {
if !self.graph.nodes.contains_key(id) {
fn ensure_node(&mut self, id: &str) -> &mut Node {
let node_defaults = &self.node_defaults;
self.graph.nodes.entry(id.to_string()).or_insert_with(|| {
let mut node = Node::new(id);
for (k, v) in &self.node_defaults {
node.attrs.insert(k.clone(), v.clone());
}
self.graph.nodes.insert(id.to_string(), node);
}
node.attrs.clone_from(node_defaults);
node
})
}
fn add_class_to_node(node: &mut Node, cls: &str) {
@ -90,12 +105,7 @@ impl SemanticState {
}
fn process_node(&mut self, node_stmt: &NodeStmt, subgraph_class: Option<&str>) {
self.ensure_node(&node_stmt.id);
let node = self
.graph
.nodes
.get_mut(&node_stmt.id)
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
let node = self.ensure_node(&node_stmt.id);
if let Some(attrs) = &node_stmt.attrs {
for (k, v) in attrs {
node.attrs.insert(k.clone(), convert_value(v));
@ -111,11 +121,6 @@ impl SemanticState {
.and_then(AttrValue::as_str)
.map(String::from);
if let Some(class_str) = class_str {
let node = self
.graph
.nodes
.get_mut(&node_stmt.id)
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
for cls in class_str.split(',') {
let cls = cls.trim().to_string();
if !cls.is_empty() && !node.classes.contains(&cls) {
@ -127,12 +132,11 @@ impl SemanticState {
fn process_edge(&mut self, edge_stmt: &EdgeStmt, subgraph_class: Option<&str>) {
for id in &edge_stmt.nodes {
self.ensure_node(id);
if !self.declared_node_ids.contains(id) {
continue;
}
let node = self.ensure_node(id);
if let Some(cls) = subgraph_class {
let node =
self.graph.nodes.get_mut(id).expect(
"node was just inserted by ensure_node, so get_mut cannot return None",
);
Self::add_class_to_node(node, cls);
}
}
@ -251,7 +255,9 @@ impl SemanticState {
///
/// Returns an error if the AST cannot be converted to a valid graph.
pub fn ast_to_graph(dot: &DotGraph) -> Result<Graph, Error> {
let mut state = SemanticState::new(dot.name.clone());
let mut declared_node_ids = HashSet::new();
collect_declared_node_ids(&dot.statements, &mut declared_node_ids);
let mut state = SemanticState::new(dot.name.clone(), declared_node_ids);
let empty = HashMap::new();
state.process_statements(&dot.statements, None, &empty, &empty);
Ok(state.graph)
@ -515,7 +521,7 @@ mod tests {
}
#[test]
fn ast_to_graph_implicit_nodes_from_edges() {
fn ast_to_graph_keeps_undeclared_edge_endpoints_out_of_nodes() {
let dot = DotGraph {
name: "Implicit".into(),
statements: vec![Statement::Edge(EdgeStmt {
@ -524,8 +530,83 @@ mod tests {
})],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.is_empty());
assert_eq!(graph.edges, vec![Edge::new("a", "b")]);
}
#[test]
fn ast_to_graph_includes_only_declared_edge_endpoints() {
let dot = DotGraph {
name: "Declared".into(),
statements: vec![
Statement::Node(NodeStmt {
id: "a".into(),
attrs: None,
}),
Statement::Edge(EdgeStmt {
nodes: vec!["a".into(), "b".into()],
attrs: None,
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("a"));
assert!(!graph.nodes.contains_key("b"));
}
#[test]
fn ast_to_graph_declaration_after_edge_still_counts() {
let dot = DotGraph {
name: "DeclaredLater".into(),
statements: vec![
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("first".into()))]),
Statement::Edge(EdgeStmt {
nodes: vec!["a".into(), "b".into()],
attrs: None,
}),
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("second".into()))]),
Statement::Node(NodeStmt {
id: "b".into(),
attrs: Some(vec![("prompt".into(), AstValue::Str("Do it".into()))]),
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("b"));
assert!(!graph.nodes.contains_key("a"));
assert_eq!(
graph.nodes["b"]
.attrs
.get("model")
.and_then(AttrValue::as_str),
Some("first")
);
}
#[test]
fn ast_to_graph_subgraph_declaration_counts() {
let dot = DotGraph {
name: "SubgraphDeclared".into(),
statements: vec![
Statement::Edge(EdgeStmt {
nodes: vec!["start".into(), "plan".into()],
attrs: None,
}),
Statement::Subgraph(SubgraphStmt {
name: Some("cluster_loop".into()),
statements: vec![Statement::Node(NodeStmt {
id: "plan".into(),
attrs: None,
})],
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("plan"));
assert!(!graph.nodes.contains_key("start"));
}
}

View file

@ -19,6 +19,7 @@ use fabro_types::{
SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection,
StageModelUsage, StageOutcome, StageProjection, StageState, StartRecord, SubAgentProjection,
SubAgentStatus, TodoListKind, TodoListProjection, TodoProjection, WorkflowRef, first_event_seq,
timing,
};
use fabro_util::error::render_compact_with_causes;
@ -405,7 +406,7 @@ impl RunProjectionReducer for RunProjection {
};
stage.response = response;
stage.completion = Some(completion);
stage.timing = Some(props.timing);
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@ -428,7 +429,7 @@ impl RunProjectionReducer for RunProjection {
failure_reason,
timestamp: ts,
});
stage.timing = Some(props.timing);
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@ -449,7 +450,7 @@ impl RunProjectionReducer for RunProjection {
context_window.event_seq = Some(event.seq);
stage.context_window = Some(context_window);
}
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentLlmStarted(props) => {
open_inference_bracket(self, stored, props, event.seq, ts);
@ -478,10 +479,10 @@ impl RunProjectionReducer for RunProjection {
inference.first_output_kind = None;
}
EventBody::AgentError(props) => {
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSessionEnded(_) => {
close_inference_brackets_for_session(self, stored);
close_active_brackets_for_session(self, stored, ts);
}
EventBody::AgentSessionActivated(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@ -497,7 +498,7 @@ impl RunProjectionReducer for RunProjection {
return Ok(());
};
stage.agent_control = AgentControlState::WaitingForSteer;
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSteeringInjected(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@ -520,12 +521,17 @@ impl RunProjectionReducer for RunProjection {
};
stage.agent_tools.clone_from(&props.tools);
}
// `AgentAcpStarted` is the start-of-process signal for an external
// ACP agent. `provider_used` is intentionally sourced from the
// subsequent `AgentSessionActivated` event, which carries the
// canonical provider/model. ACP runs without a steering hub never
// emit activation and so legitimately leave `provider_used`
// unset — matching legacy ACP behavior.
EventBody::AgentAcpStarted(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
};
stage.open_acp_inference(ts);
// `provider_used` is intentionally sourced from the subsequent
// `AgentSessionActivated` event, which carries the canonical
// provider/model. ACP runs without a steering hub never emit
// activation and legitimately leave it unset.
}
EventBody::CommandStarted(props) => {
let script_invocation = serde_json::to_value(props).map_err(|err| {
Error::InvalidEvent(format!("invalid command.started payload: {err}"))
@ -552,6 +558,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -564,6 +571,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -576,6 +584,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -594,6 +603,11 @@ impl RunProjectionReducer for RunProjection {
// Branches bypass the engine's StageStarted/StageCompleted
// lifecycle. Seed started_at so the branch stage drives a live
// wall-clock timer while it runs (the entry is created Running).
let handler = stored
.node_id
.as_deref()
.and_then(|node_id| self.spec().graph.nodes.get(node_id))
.map(|node| StageHandler::from_handler_type(node.handler_type()));
let is_new = stored
.stage_id
.as_ref()
@ -604,6 +618,9 @@ impl RunProjectionReducer for RunProjection {
if stage.started_at.is_none() {
stage.started_at = Some(ts);
}
if stage.handler.is_none() {
stage.handler = handler;
}
if is_new {
stage.graph_visit = props.graph_visit;
stage
@ -746,6 +763,11 @@ impl RunProjectionReducer for RunProjection {
});
}
EventBody::AgentToolStarted(props) => {
let root_session_id = if stored.parent_session_id.is_none() {
stored.session_id.clone()
} else {
None
};
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
@ -766,6 +788,25 @@ impl RunProjectionReducer for RunProjection {
projection.invoked = true;
}
}
// A subagent's tools run inside the root session's tool call,
// so the root batch already covers them. Timing them again
// would double-count that span.
if let Some(session_id) = root_session_id {
stage.open_tool_call(session_id, props.tool_call_id.clone(), ts);
}
}
EventBody::AgentToolCompleted(props) => {
if stored.parent_session_id.is_some() {
return Ok(());
}
let Some(session_id) = stored.session_id.as_deref() else {
return Ok(());
};
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
};
stage.close_tool_call(session_id, &props.tool_call_id, ts);
}
_ => {}
}
@ -1031,19 +1072,19 @@ fn open_inference_bracket(
});
}
/// Resolve the stage's `inference` slot when it holds a bracket this event is
/// allowed to mutate.
/// Resolve the stage owning a bracket this event is allowed to mutate,
/// borrowing the whole projection so the caller can also fold elapsed time
/// into the stage's live accumulators.
///
/// `None` when the event came from a child session, when the stage has no
/// open bracket, or when the bracket belongs to a different session — the
/// last case matters after failover, which discards the session and builds a
/// new one within a single visit.
fn matching_inference_slot<'a>(
/// Returns `None` for child-session events, for a stage with no open bracket,
/// and for a bracket belonging to a different session (which is what
/// post-failover events look like).
fn matching_inference_stage<'a>(
state: &'a mut RunProjection,
stored: &RunEvent,
visit: u32,
seq: u32,
) -> Option<&'a mut Option<StageInferenceProjection>> {
) -> Option<&'a mut StageProjection> {
if stored.parent_session_id.is_some() {
return None;
}
@ -1053,7 +1094,7 @@ fn matching_inference_slot<'a>(
.inference
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
opened_here.then_some(&mut stage.inference)
opened_here.then_some(stage)
}
/// Resolve the open inference bracket this event is allowed to mutate.
@ -1063,17 +1104,39 @@ fn matching_inference_bracket<'a>(
visit: u32,
seq: u32,
) -> Option<&'a mut StageInferenceProjection> {
matching_inference_slot(state, stored, visit, seq)?.as_mut()
matching_inference_stage(state, stored, visit, seq)?
.inference
.as_mut()
}
/// Close the bracket on a stage-addressed terminal event.
fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: u32, seq: u32) {
if let Some(slot) = matching_inference_slot(state, stored, visit, seq) {
*slot = None;
}
/// Close the bracket on a stage-addressed terminal event, folding its elapsed
/// time into the stage's live inference accumulator.
fn close_inference_bracket(
state: &mut RunProjection,
stored: &RunEvent,
visit: u32,
seq: u32,
ts: DateTime<Utc>,
) {
let Some(stage) = matching_inference_stage(state, stored, visit, seq) else {
return;
};
close_bracket_on_stage(stage, ts);
}
/// Close every bracket opened by the session that just ended.
/// Take the open bracket and add its span to `live_inference_ms`.
///
/// Retries inside the bracket are deliberately included: the in-process
/// stopwatch counts a retried attempt's elapsed time as inference, and
/// `agent.llm.retry` keeps the bracket open rather than reopening it.
fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime<Utc>) {
let Some(inference) = stage.inference.take() else {
return;
};
stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, ts));
}
/// Close every active bracket opened by the session that just ended.
///
/// `agent.session.ended` is the only ordering-safe backstop for terminal
/// cancel and wall-clock timeout, which tear the session down through
@ -1088,7 +1151,11 @@ fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit:
/// opened. Implemented as a normal stage lookup it would find no target and
/// silently no-op, leaving the bracket open forever on exactly the path it
/// exists to cover.
fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunEvent) {
fn close_active_brackets_for_session(
state: &mut RunProjection,
stored: &RunEvent,
ts: DateTime<Utc>,
) {
if stored.parent_session_id.is_some() {
return;
}
@ -1101,8 +1168,9 @@ fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunE
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
if opened_here {
stage.inference = None;
close_bracket_on_stage(stage, ts);
}
stage.close_tool_batch_for_session(session_id, ts);
}
}
@ -1412,23 +1480,26 @@ fn finalize_unfinished_stages_after_run_failed(
StageState::Failed
};
for (_, stage) in state.iter_stages_mut() {
for (_, stage) in state.iter_stages_unordered_mut() {
if stage.state.is_terminal() {
continue;
}
// Close any brackets still open so their spans are not dropped on the
// floor when the live estimate is frozen into `timing` below.
close_bracket_on_stage(stage, timestamp);
stage.close_open_acp_inference(timestamp);
stage.close_open_tool_batch(timestamp);
// Freeze the live estimate before flipping to a terminal state:
// `live_timing` reads `effective_state` and would return wall-only
// once the stage no longer looks in-flight.
let frozen = stage.live_timing(timestamp);
stage.state = terminal_state;
if stage.timing.is_none() {
if let Some(started_at) = stage.started_at {
let wall_time_ms = u64::try_from(
timestamp
.signed_duration_since(started_at)
.num_milliseconds()
.max(0),
)
.expect("non-negative milliseconds fit in u64");
stage.timing = Some(fabro_types::StageTiming::wall_only(wall_time_ms));
}
if stage.timing.is_none() && stage.started_at.is_some() {
stage.set_authoritative_timing(frozen);
} else {
stage.clear_live_timing();
}
}
}
@ -1545,21 +1616,519 @@ mod tests {
};
use fabro_types::settings::run::{DockerfileSource, EnvironmentProvider};
use fabro_types::{
AgentBackend, AgentControlState, AutomationRef, BilledModelUsage, BilledTokenCounts,
BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody,
FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Outcome,
PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
AgentBackend, AgentControlState, AttrValue, AutomationRef, BilledModelUsage,
BilledTokenCounts, BlockedReason, Checkpoint, CheckpointRecord, CommandTermination,
EventBody, FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Node,
Outcome, PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
RunApprovalState, RunBlobId, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec,
RunStatus, Speed, StageContextWindowBreakdownItem, StageContextWindowCategory,
StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
StageContextWindowWarning, StageModelUsage, StageOutcome, StageState, SubAgentStatus,
SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support,
StageContextWindowWarning, StageHandler, StageModelUsage, StageOutcome, StageState,
StageTiming, SubAgentStatus, SuccessReason, WorkflowSettings, first_event_seq, fixtures,
test_support,
};
use serde_json::json;
use super::{RunProjection, RunProjectionReducer, build_summary};
use crate::{Error, EventEnvelope, StageId};
/// Live accumulation of inference and tool time while a stage is in
/// flight. The finalized breakdown still arrives with the terminal event
/// and replaces these; these exist so a long-running stage is not reported
/// as doing no work.
mod live_active_accumulation {
use fabro_types::run_event::{
AgentLlmFirstOutputProps, AgentLlmRetryProps, AgentLlmStartedProps,
AgentToolCompletedProps, AgentToolStartedProps,
};
use fabro_types::{
LlmOutputKind, LlmRetryPhase, ModelRef, Speed, StageOutcome, StageProjection,
};
use super::*;
fn stage_id() -> StageId {
StageId::new("plan", 1)
}
fn session_event(seq: u32, ts: &str, session_id: &str, body: EventBody) -> EventEnvelope {
let mut event = test_stage_event_at(seq, ts, body, stage_id());
event.event.session_id = Some(session_id.to_string());
event
}
fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
session_event(seq, ts, "session-1", body)
}
/// An event from a sub-agent session nested under the root session.
fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
let mut event = agent_event(seq, ts, body);
event.event.session_id = Some("session-child".to_string());
event.event.parent_session_id = Some("session-1".to_string());
event
}
fn llm_started() -> EventBody {
EventBody::AgentLlmStarted(AgentLlmStartedProps {
requested_model: ModelRef {
provider: "anthropic".parse().unwrap(),
model_id: "claude-fable-5".into(),
speed: Some(Speed::Fast),
},
visit: 1,
})
}
fn tool_started(tool_call_id: &str) -> EventBody {
EventBody::AgentToolStarted(AgentToolStartedProps {
tool_name: "Bash".to_string(),
tool_call_id: tool_call_id.to_string(),
arguments: json!({}),
visit: 1,
tool_call: None,
turn_id: None,
parent_message_id: None,
})
}
fn tool_completed(tool_call_id: &str) -> EventBody {
EventBody::AgentToolCompleted(AgentToolCompletedProps {
tool_name: "Bash".to_string(),
tool_call_id: tool_call_id.to_string(),
output: json!("ok"),
is_error: false,
visit: 1,
tool_result: None,
turn_id: None,
})
}
fn agent_message() -> EventBody {
EventBody::AgentMessage(live_agent_message_props(live_counts(10, 5)))
}
fn started_state() -> RunProjection {
let mut state = initialized_projection();
state
.apply_event(&test_stage_event_at(
1,
"2026-04-07T12:00:00Z",
EventBody::StageStarted(started_props()),
stage_id(),
))
.unwrap();
state
}
fn stage(state: &RunProjection) -> &StageProjection {
state.stage(&stage_id()).unwrap()
}
#[test]
fn closing_an_inference_bracket_accumulates_its_span() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:06Z",
EventBody::AgentLlmFirstOutput(AgentLlmFirstOutputProps {
kind: LlmOutputKind::Text,
visit: 1,
}),
))
.unwrap();
state
.apply_event(&agent_event(4, "2026-04-07T12:00:12Z", agent_message()))
.unwrap();
// 12:00:05 -> 12:00:12; first_output is a marker, not the close.
assert_eq!(stage(&state).live_inference_ms, 7_000);
assert!(stage(&state).inference.is_none());
}
#[test]
fn concurrent_tool_calls_count_once_not_per_call() {
let mut state = started_state();
for (seq, id) in [(2, "call-a"), (3, "call-b"), (4, "call-c")] {
state
.apply_event(&agent_event(seq, "2026-04-07T12:00:00Z", tool_started(id)))
.unwrap();
}
// All three finish 10s later. Summing per-call spans would report
// 30s; the batch actually occupied 10s of wall time.
for (seq, id) in [(5, "call-a"), (6, "call-b"), (7, "call-c")] {
state
.apply_event(&agent_event(
seq,
"2026-04-07T12:00:10Z",
tool_completed(id),
))
.unwrap();
}
assert_eq!(stage(&state).live_tool_ms, 10_000);
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn a_batch_stays_open_until_its_last_call_reports() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:02Z",
tool_started("call-b"),
))
.unwrap();
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:05Z",
tool_completed("call-a"),
))
.unwrap();
assert_eq!(
stage(&state).live_tool_ms,
0,
"batch must not close while call-b is outstanding"
);
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:09Z",
tool_completed("call-b"),
))
.unwrap();
// Measured from the batch open, not from the last call's start.
assert_eq!(stage(&state).live_tool_ms, 9_000);
}
#[test]
fn successive_batches_accumulate() {
let mut state = started_state();
for (seq, ts, body) in [
(2, "2026-04-07T12:00:00Z", tool_started("call-a")),
(3, "2026-04-07T12:00:04Z", tool_completed("call-a")),
(4, "2026-04-07T12:00:10Z", tool_started("call-b")),
(5, "2026-04-07T12:00:16Z", tool_completed("call-b")),
] {
state.apply_event(&agent_event(seq, ts, body)).unwrap();
}
assert_eq!(stage(&state).live_tool_ms, 10_000);
}
#[test]
fn a_duplicate_completion_does_not_drain_the_batch_early() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:00Z",
tool_started("call-b"),
))
.unwrap();
// call-a reports twice, as a replayed or duplicated log can.
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:03Z",
tool_completed("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:04Z",
tool_completed("call-a"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 0);
assert!(stage(&state).tool_batch.is_some());
}
#[test]
fn a_foreign_session_completion_does_not_mutate_the_open_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&session_event(
3,
"2026-04-07T12:00:05Z",
"session-2",
tool_completed("call-a"),
))
.unwrap();
let batch = stage(&state).tool_batch.as_ref().unwrap();
assert_eq!(batch.session_id, "session-1");
assert!(batch.open_call_ids.contains("call-a"));
assert_eq!(stage(&state).live_tool_ms, 0);
}
#[test]
fn a_replacement_session_starts_a_separate_tool_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("old-call"),
))
.unwrap();
state
.apply_event(&session_event(
3,
"2026-04-07T12:00:05Z",
"session-2",
tool_started("new-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 5_000);
let batch = stage(&state).tool_batch.as_ref().unwrap();
assert_eq!(batch.session_id, "session-2");
assert_eq!(
batch.open_call_ids,
["new-call".to_string()].into_iter().collect()
);
// A delayed completion from the old session cannot close the new
// session's batch even when call ids happen to collide.
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:07Z",
tool_completed("new-call"),
))
.unwrap();
assert!(stage(&state).tool_batch.is_some());
state
.apply_event(&session_event(
5,
"2026-04-07T12:00:09Z",
"session-2",
tool_completed("new-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 9_000);
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn subagent_tool_calls_do_not_double_count_against_the_root_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("root-call"),
))
.unwrap();
// The sub-agent's own tools run inside the root call's span.
state
.apply_event(&child_event(
3,
"2026-04-07T12:00:01Z",
tool_started("child-call"),
))
.unwrap();
state
.apply_event(&child_event(
4,
"2026-04-07T12:00:02Z",
tool_completed("child-call"),
))
.unwrap();
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:08Z",
tool_completed("root-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 8_000);
}
#[test]
fn session_end_accumulates_every_open_active_bracket() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:07Z",
tool_started("call-a"),
))
.unwrap();
let mut ended = test_stage_event_at(
4,
"2026-04-07T12:00:20Z",
EventBody::AgentSessionEnded(AgentSessionEndedProps {}),
stage_id(),
);
ended.event.session_id = Some("session-1".to_string());
state.apply_event(&ended).unwrap();
assert_eq!(stage(&state).live_inference_ms, 15_000);
assert_eq!(stage(&state).live_tool_ms, 13_000);
assert!(stage(&state).inference.is_none());
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn a_foreign_session_close_leaves_the_bracket_and_accumulator_alone() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
// Post-failover: a new session emits the message, so the old
// bracket is not this event's to close or bill.
let mut foreign =
test_stage_event_at(3, "2026-04-07T12:00:20Z", agent_message(), stage_id());
foreign.event.session_id = Some("session-2".to_string());
state.apply_event(&foreign).unwrap();
assert_eq!(stage(&state).live_inference_ms, 0);
assert!(stage(&state).inference.is_some());
}
#[test]
fn a_retry_keeps_accumulating_within_one_bracket() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:04Z",
EventBody::AgentLlmRetry(AgentLlmRetryProps {
provider: "anthropic".to_string(),
model: "claude-fable-5".to_string(),
attempt: 0,
delay_secs: 0.0,
error: json!({ "kind": "stream" }),
phase: Some(LlmRetryPhase::Consume),
visit: 1,
}),
))
.unwrap();
state
.apply_event(&agent_event(4, "2026-04-07T12:00:11Z", agent_message()))
.unwrap();
// The whole bracket counts, retry included, matching the
// in-process stopwatch.
assert_eq!(stage(&state).live_inference_ms, 11_000);
assert_eq!(stage(&state).inference, None);
}
#[test]
fn stage_completion_replaces_the_live_estimate_with_finalized_timing() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(3, "2026-04-07T12:00:09Z", agent_message()))
.unwrap();
assert_eq!(stage(&state).live_inference_ms, 9_000);
state
.apply_event(&test_stage_event_at(
4,
"2026-04-07T12:00:10Z",
EventBody::StageCompleted(completed_props(10_000, StageOutcome::Succeeded)),
stage_id(),
))
.unwrap();
let stage = stage(&state);
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:30:00Z")),
stage.timing.unwrap(),
"a terminal stage reports its finalized breakdown, not a live estimate"
);
assert_eq!(stage.live_inference_ms, 0);
assert_eq!(stage.live_tool_ms, 0);
assert!(stage.inference.is_none());
assert!(stage.tool_batch.is_none());
}
#[test]
fn run_failure_freezes_open_work_and_clears_live_bookkeeping() {
let mut state = started_state();
state.status = RunStatus::Running;
state
.apply_event(&agent_event(2, "2026-04-07T12:00:01Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(3, "2026-04-07T12:00:04Z", agent_message()))
.unwrap();
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:05Z",
tool_started("call-a"),
))
.unwrap();
let mut failed = test_event(
5,
EventBody::RunFailed(run_failed_props(FailureReason::WorkflowError)),
None,
);
failed.event.ts = test_dt("2026-04-07T12:00:10Z");
state.apply_event(&failed).unwrap();
let stage = stage(&state);
assert_eq!(
stage.timing,
Some(fabro_types::StageTiming::new(10_000, 3_000, 5_000))
);
assert_eq!(stage.state, StageState::Failed);
assert_eq!(stage.live_inference_ms, 0);
assert_eq!(stage.live_tool_ms, 0);
assert!(stage.inference.is_none());
assert!(stage.tool_batch.is_none());
}
}
fn test_event(seq: u32, body: EventBody, node_id: Option<&str>) -> EventEnvelope {
let event = RunEvent {
id: format!("evt-{seq}"),
@ -2365,6 +2934,49 @@ mod tests {
);
}
#[test]
fn parallel_branch_started_uses_the_graph_handler_for_live_timing() {
for (handler_type, handler, expected) in [
(
"prompt",
StageHandler::Prompt,
StageTiming::new(5_000, 5_000, 0),
),
(
"command",
StageHandler::Command,
StageTiming::new(5_000, 0, 5_000),
),
] {
let mut spec = test_run_spec();
let mut node = Node::new("review");
node.attrs.insert(
"type".to_string(),
AttrValue::String(handler_type.to_string()),
);
spec.graph.nodes.insert(node.id.clone(), node);
let mut state = RunProjection::new("Test run".to_string(), spec, Utc::now());
let branch = StageId::new("review", 1);
state
.apply_event(&test_stage_event_at(
3,
"2026-04-07T12:00:00Z",
EventBody::ParallelBranchStarted(ParallelBranchStartedProps {
index: 0,
graph_visit: None,
resumed_from_stage_id: None,
}),
branch.clone(),
))
.unwrap();
let stage = state.stage(&branch).unwrap();
assert_eq!(stage.handler, Some(handler));
assert_eq!(stage.live_timing(test_dt("2026-04-07T12:00:05Z")), expected);
}
}
fn start_stage(state: &mut RunProjection, stage_id: &StageId) {
state
.apply_event(&test_stage_event(
@ -2509,6 +3121,66 @@ mod tests {
assert_eq!(provider_used.model.as_deref(), Some("fake"));
}
#[test]
fn acp_events_accumulate_live_inference_time() {
let mut state = initialized_projection();
let stage_id = StageId::new("code", 1);
state
.apply_event(&test_stage_event_at(
3,
"2026-04-07T12:00:00Z",
EventBody::StageStarted(StageStartedProps {
graph_visit: None,
resumed_from_stage_id: None,
index: 0,
handler_type: "agent".to_string(),
attempt: 1,
max_attempts: 1,
}),
stage_id.clone(),
))
.unwrap();
state
.apply_event(&test_stage_event_at(
4,
"2026-04-07T12:00:05Z",
EventBody::AgentAcpStarted(AgentAcpStartedProps {
visit: 1,
command: "python fake_agent.py".to_string(),
config_name: Some("fake".to_string()),
}),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:00:15Z")),
StageTiming::new(15_000, 10_000, 0)
);
state
.apply_event(&test_stage_event_at(
5,
"2026-04-07T12:00:17Z",
EventBody::AgentAcpCompleted(AgentAcpCompletedProps {
stdout: "done".to_string(),
stderr: String::new(),
stop_reason: "end_turn".to_string(),
duration_ms: 12_000,
}),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:00:20Z")),
StageTiming::new(20_000, 12_000, 0)
);
}
#[test]
fn agent_acp_completed_updates_stage_output_projection() {
let mut state = initialized_projection();

View file

@ -3,7 +3,7 @@ use std::fmt::Write as _;
use std::sync::LazyLock;
use chrono::{DateTime, Utc};
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming};
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming, timing};
use sqlx::sqlite::{SqliteConnection, SqliteRow};
use sqlx::{QueryBuilder, Row as _, Sqlite, SqlitePool};
use strum::VariantArray as _;
@ -545,7 +545,7 @@ fn overlay_live_wall_time(run: &mut Run, now: DateTime<Utc>) {
let Some(started_at) = run.timestamps.started_at else {
return;
};
let wall_time_ms = RunTiming::wall_time_ms_since(started_at, now);
let wall_time_ms = timing::elapsed_ms(started_at, now);
run.timing = Some(
run.timing
.unwrap_or_else(|| RunTiming::wall_only(wall_time_ms))

View file

@ -327,6 +327,18 @@ impl Database {
Ok(self.projection_cache.get(run_id).await)
}
pub async fn get_cached_projection(
&self,
run_id: &RunId,
) -> Result<Option<Arc<RunProjection>>> {
self.warm_projection_cache().await?;
Ok(self
.projection_cache
.projection_snapshot(run_id)
.await
.map(|(projection, _)| projection))
}
pub async fn get_cached_summary(
&self,
run_id: &RunId,

View file

@ -1,3 +1,5 @@
use std::collections::HashSet;
use fabro_graphviz::graph::Graph;
use crate::{Diagnostic, LintRule, Severity};
@ -8,6 +10,26 @@ pub(super) fn rule() -> Box<dyn LintRule> {
struct Rule;
impl Rule {
fn diagnostic(&self, node_id: &str, from: &str, to: &str) -> Diagnostic {
Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!(
"Node '{node_id}' is referenced by edge '{from} -> {to}' but has no node \
declaration"
),
node_id: Some(node_id.to_string()),
edge: Some((from.to_string(), to.to_string())),
fix: Some(format!(
"Declare node '{node_id}' or correct the edge endpoint"
)),
..Diagnostic::default()
}
}
}
impl LintRule for Rule {
fn name(&self) -> &'static str {
"edge_target_exists"
@ -15,36 +37,12 @@ impl LintRule for Rule {
fn apply(&self, graph: &Graph) -> Vec<Diagnostic> {
let mut diagnostics = Vec::new();
let mut reported = HashSet::new();
for edge in &graph.edges {
if !graph.nodes.contains_key(&edge.to) {
diagnostics.push(Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!(
"Edge from '{}' targets non-existent node '{}'",
edge.from, edge.to
),
node_id: None,
edge: Some((edge.from.clone(), edge.to.clone())),
fix: Some(format!("Define node '{}' or fix the edge target", edge.to)),
..Diagnostic::default()
});
}
if !graph.nodes.contains_key(&edge.from) {
diagnostics.push(Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!("Edge source '{}' references non-existent node", edge.from),
node_id: None,
edge: Some((edge.from.clone(), edge.to.clone())),
fix: Some(format!(
"Define node '{}' or fix the edge source",
edge.from
)),
..Diagnostic::default()
});
for endpoint in [&edge.to, &edge.from] {
if !graph.nodes.contains_key(endpoint) && reported.insert(endpoint) {
diagnostics.push(self.diagnostic(endpoint, &edge.from, &edge.to));
}
}
}
diagnostics
@ -53,11 +51,112 @@ impl LintRule for Rule {
#[cfg(test)]
mod tests {
use fabro_graphviz::graph::Edge;
use fabro_graphviz::graph::{Edge, Graph};
use fabro_graphviz::parser;
use super::Rule;
use crate::rules::test_support::minimal_graph;
use crate::{LintRule, Severity};
use crate::{Diagnostic, LintRule, Severity};
fn parse(dot: &str) -> Graph {
parser::parse(dot).expect("fixture should parse")
}
fn undeclared_nodes(graph: &Graph) -> Vec<String> {
Rule.apply(graph)
.into_iter()
.map(|d| d.node_id.expect("diagnostic should name a node"))
.collect()
}
#[test]
fn edge_only_node_is_rejected() {
let graph = parse(
r"digraph EdgeOnly {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> misspelled_node
misspelled_node -> exit
}",
);
let diagnostics = Rule.apply(&graph);
assert_eq!(diagnostics.len(), 1, "diagnostics: {diagnostics:?}");
let Diagnostic {
severity,
node_id,
edge,
..
} = &diagnostics[0];
assert_eq!(*severity, Severity::Error);
assert_eq!(node_id.as_deref(), Some("misspelled_node"));
assert_eq!(
edge.clone(),
Some(("start".to_string(), "misspelled_node".to_string()))
);
}
#[test]
fn declaration_after_the_edge_is_accepted() {
let graph = parse(
r#"digraph DeclaredLater {
start -> work
work [prompt="Do the work"]
work -> exit
start [shape=Mdiamond]
exit [shape=Msquare]
}"#,
);
assert!(Rule.apply(&graph).is_empty());
}
#[test]
fn chained_edges_report_every_undeclared_endpoint() {
let graph = parse(
r"digraph Chained {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> first -> second -> exit
}",
);
assert_eq!(undeclared_nodes(&graph), vec!["first", "second"]);
}
#[test]
fn a_node_is_reported_once_no_matter_how_many_edges_use_it() {
let graph = parse(
r"digraph Repeated {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> typo
typo -> exit
typo -> start
}",
);
assert_eq!(undeclared_nodes(&graph), vec!["typo"]);
}
#[test]
fn subgraph_declaration_is_accepted() {
let graph = parse(
r#"digraph Subgraphed {
start [shape=Mdiamond]
exit [shape=Msquare]
subgraph cluster_loop {
label = "Loop A"
plan [prompt="Plan the work"]
}
start -> plan -> exit
}"#,
);
assert!(Rule.apply(&graph).is_empty());
}
#[test]
fn edge_target_exists_rule_missing_target() {

View file

@ -1,32 +1,7 @@
use std::borrow::Cow;
use std::collections::HashMap;
use fabro_model::Catalog;
use fabro_types::{
BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageTiming,
};
fn stage_usage_with_cost<'a>(
catalog: Option<&Catalog>,
stage: &'a StageProjection,
) -> Cow<'a, BilledTokenCounts> {
let Some(catalog) = catalog else {
return Cow::Borrowed(&stage.usage);
};
let Some(model) = stage.model.as_ref() else {
return Cow::Borrowed(&stage.usage);
};
if stage.usage.total_usd_micros.is_some() {
return Cow::Borrowed(&stage.usage);
}
let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
return Cow::Borrowed(&stage.usage);
};
let mut usage = stage.usage.clone();
usage.total_usd_micros = Some(total_usd_micros);
Cow::Owned(usage)
}
use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageTiming};
#[derive(Debug, Clone, PartialEq)]
pub struct ProjectionBillingStage {
@ -80,7 +55,7 @@ pub fn billing_rollup_from_projection(
if is_boundary_stage(projection, stage_id.node_id()) {
continue;
}
let usage = stage_usage_with_cost(catalog, stage);
let usage = stage.billed_usage(catalog);
let usage = usage.as_ref();
if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() {
continue;

View file

@ -16,7 +16,7 @@ use crate::artifact_upload::ArtifactSink;
use crate::condition::evaluate_condition;
use crate::context::{Context, WorkflowContext, context_diff_public, keys};
use crate::error::Error;
use crate::operations::{ValidateInput, WorkflowInput, validate};
use crate::operations::{ValidateInput, WorkflowInput, validate_with_catalog};
use crate::outcome::{Outcome, OutcomeExt, StageOutcome};
use crate::pipeline::types::Initialized;
use crate::run_options::RunOptions;
@ -65,20 +65,14 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
.get("stack.child_dot_source")
.and_then(|v| v.as_str())
{
let mut validated = validate(ValidateInput {
workflow: WorkflowInput::DotSource {
let graph = validate_child_workflow(
WorkflowInput::DotSource {
source: dot.to_string(),
base_dir: None,
},
settings: WorkflowSettings::default(),
vars: std::collections::HashMap::new(),
cwd: cwd.clone(),
custom_transforms: Vec::new(),
catalog: Arc::clone(&services.run.catalog),
})?;
validated.promote_template_undefined_variables_to_errors();
validated.raise_on_errors()?;
let (graph, _, _) = validated.into_parts();
cwd,
services,
)?;
return Ok(ParsedChildWorkflow {
graph,
workflow_path: None,
@ -113,17 +107,7 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
WorkflowInput::Bundled(workflow) => Some(workflow.path.clone()),
WorkflowInput::Path(_) | WorkflowInput::DotSource { .. } => None,
};
let mut validated = validate(ValidateInput {
workflow,
settings: WorkflowSettings::default(),
vars: std::collections::HashMap::new(),
cwd,
custom_transforms: Vec::new(),
catalog: Arc::clone(&services.run.catalog),
})?;
validated.promote_template_undefined_variables_to_errors();
validated.raise_on_errors()?;
let (graph, _, _) = validated.into_parts();
let graph = validate_child_workflow(workflow, cwd, services)?;
return Ok(ParsedChildWorkflow {
graph,
workflow_path,
@ -132,6 +116,29 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
Err(Error::handler("No child workflow source".to_string()))
}
/// Validate a child workflow against the run's catalog, failing on any error
/// diagnostic (undefined template variables included).
fn validate_child_workflow(
workflow: WorkflowInput,
cwd: PathBuf,
services: &EngineServices,
) -> Result<Graph, Error> {
let mut validated = validate_with_catalog(
ValidateInput {
workflow,
settings: WorkflowSettings::default(),
vars: HashMap::new(),
cwd,
custom_transforms: Vec::new(),
},
Arc::clone(&services.run.catalog),
)?;
validated.promote_template_undefined_variables_to_errors();
validated.raise_on_errors()?;
let (graph, _, _) = validated.into_parts();
Ok(graph)
}
#[async_trait]
impl Handler for SubWorkflowHandler {
async fn execute(

View file

@ -28,7 +28,7 @@ use crate::pipeline::{self, Persisted, TransformOptions, Validated};
use crate::records::RunSpec;
use crate::run_lookup::default_scratch_base;
use crate::run_materialization::materialize_run;
use crate::transforms::{RenderMode, Transform};
use crate::transforms::{ModelResolutionTransform, RenderMode};
use crate::workflow_bundle::{RunDefinition, WorkflowBundle};
#[derive(Clone, Debug)]
@ -293,28 +293,21 @@ fn create_from_source(
file_resolver: Option<Arc<dyn FileResolver>>,
goal_override: Option<&str>,
) -> Result<Persisted, Error> {
let template_context = template_context(Some(&options.settings), vars);
let mut validated = preprocess_and_validate(
dot_source,
options.source_name.clone(),
let mut validated = preprocess_and_validate(dot_source, goal_override, &TransformOptions {
current_dir,
file_resolver,
Vec::new(),
template_context,
goal_override,
RenderMode::Structural,
options
.settings
.run
.model
.provider
.as_deref()
.filter(|provider| !provider.is_empty())
.map(ProviderId::new),
&options.configured_providers,
false,
&options.catalog,
)?;
template_context: template_context(Some(&options.settings), vars),
source_name: options.source_name.clone(),
render_mode: RenderMode::Structural,
custom_transforms: Vec::new(),
model_resolution: Some(
ModelResolutionTransform::for_eligible(
Arc::clone(&options.catalog),
options.configured_providers.iter().cloned().collect(),
)
.with_default_provider(configured_default_provider(&options.settings)),
),
})?;
validated.promote_template_undefined_variables_to_errors();
if validated.has_errors() {
@ -326,36 +319,37 @@ fn create_from_source(
persist_validated(validated, options)
}
/// Parse, transform, and validate `dot_source`.
///
/// `options.model_resolution` drives both halves of catalog awareness: it
/// selects concrete models during TRANSFORM and enables the catalog-backed
/// lint rules during VALIDATE. `None` leaves authored model and provider
/// selectors untouched for offline structural validation.
pub(super) fn preprocess_and_validate(
dot_source: &str,
source_name: Option<String>,
current_dir: Option<PathBuf>,
file_resolver: Option<Arc<dyn FileResolver>>,
custom_transforms: Vec<Box<dyn Transform>>,
template_context: TemplateContext,
goal_override: Option<&str>,
render_mode: RenderMode,
default_provider: Option<ProviderId>,
eligible_providers: &[ProviderId],
catalog_fallback: bool,
catalog: &Arc<Catalog>,
options: &TransformOptions,
) -> Result<Validated, Error> {
let mut parsed = pipeline::parse(dot_source)?;
apply_goal_override(&mut parsed.graph, goal_override);
let transformed = pipeline::transform(parsed, &TransformOptions {
current_dir,
file_resolver,
template_context,
source_name,
render_mode,
custom_transforms,
catalog: Arc::clone(catalog),
default_provider,
eligible_providers: eligible_providers.iter().cloned().collect(),
catalog_fallback,
})?;
Ok(pipeline::validate(transformed, catalog.as_ref(), &[]))
let transformed = pipeline::transform(parsed, options)?;
let catalog = options
.model_resolution
.as_ref()
.map(ModelResolutionTransform::catalog);
Ok(pipeline::validate(transformed, catalog, &[]))
}
/// The workflow-level default provider, treating an empty setting as unset.
pub(super) fn configured_default_provider(settings: &WorkflowSettings) -> Option<ProviderId> {
settings
.run
.model
.provider
.as_deref()
.filter(|provider| !provider.is_empty())
.map(ProviderId::new)
}
pub(super) fn template_context(
@ -462,8 +456,9 @@ mod tests {
use object_store::memory::InMemory;
use super::*;
use crate::operations::{ValidateInput, validate};
use crate::operations::{ValidateInput, validate, validate_with_catalog};
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
use crate::transforms::Transform;
use crate::workflow_bundle::BundledWorkflow;
fn memory_store() -> Arc<Database> {
Arc::new(Database::new(
@ -553,17 +548,19 @@ reasoning = false
}
fn validate_dot(dot_source: &str, settings: WorkflowSettings) -> Validated {
validate(ValidateInput {
workflow: WorkflowInput::DotSource {
source: dot_source.to_string(),
base_dir: None,
validate_with_catalog(
ValidateInput {
workflow: WorkflowInput::DotSource {
source: dot_source.to_string(),
base_dir: None,
},
settings,
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
},
settings,
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
test_catalog(),
)
.unwrap()
}
@ -573,21 +570,35 @@ reasoning = false
fn validate_dot_with_vars(dot_source: &str, vars: HashMap<String, String>) -> Validated {
preprocess_and_validate(
dot_source,
Some("workflow.fabro".to_string()),
Some(PathBuf::from(".")),
None,
Vec::new(),
template_context(Some(&WorkflowSettings::default()), vars),
None,
RenderMode::Structural,
None,
&test_provider_ids(),
false,
&test_catalog(),
&test_transform_options(
PathBuf::from("."),
None,
RenderMode::Structural,
template_context(Some(&WorkflowSettings::default()), vars),
),
)
.unwrap()
}
/// Catalog-backed TRANSFORM options for the built-in test catalog.
fn test_transform_options(
current_dir: PathBuf,
file_resolver: Option<Arc<dyn FileResolver>>,
render_mode: RenderMode,
template_context: TemplateContext,
) -> TransformOptions {
TransformOptions {
current_dir: Some(current_dir),
file_resolver,
template_context,
source_name: Some("workflow.fabro".to_string()),
render_mode,
custom_transforms: Vec::new(),
model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
}
}
const MINIMAL_DOT: &str = r#"digraph Test {
graph [goal="Build feature"]
start [shape=Mdiamond]
@ -744,17 +755,13 @@ reasoning = false
let result = preprocess_and_validate(
dot,
Some("workflow.fabro".to_string()),
Some(PathBuf::from(".")),
None,
Vec::new(),
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
None,
RenderMode::Strict,
None,
&test_provider_ids(),
false,
&test_catalog(),
&test_transform_options(
PathBuf::from("."),
None,
RenderMode::Strict,
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
),
);
let Err(err) = result else {
panic!("expected strict mode to hard-fail on unbound inline prompt");
@ -781,19 +788,15 @@ reasoning = false
let result = preprocess_and_validate(
dot,
Some("workflow.fabro".to_string()),
Some(dir.path().to_path_buf()),
Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
None,
))),
Vec::new(),
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
None,
RenderMode::Strict,
None,
&test_provider_ids(),
false,
&test_catalog(),
&test_transform_options(
dir.path().to_path_buf(),
Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
None,
))),
RenderMode::Strict,
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
),
);
let Err(err) = result else {
panic!("expected strict mode to hard-fail on unbound imported prompt");
@ -852,7 +855,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
catalog: test_catalog(),
});
assert!(result.is_err());
@ -901,7 +903,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
let file_missing = validate(ValidateInput {
@ -920,7 +921,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
assert_eq!(
@ -944,7 +944,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
let file_goal = validate(ValidateInput {
@ -963,7 +962,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
assert_eq!(
@ -1055,7 +1053,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
catalog: test_catalog(),
});
assert!(result.is_err());
}
@ -1100,7 +1097,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: vec![Box::new(TagTransform)],
catalog: test_catalog(),
})
.unwrap();
validated.raise_on_errors().unwrap();
@ -1135,7 +1131,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
validated.raise_on_errors().unwrap();
@ -1177,7 +1172,6 @@ reasoning = false
vars: HashMap::new(),
cwd: dir.path().to_path_buf(),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
@ -1227,7 +1221,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();
@ -1278,7 +1271,6 @@ reasoning = false
vars: HashMap::new(),
cwd: PathBuf::from("."),
custom_transforms: Vec::new(),
catalog: test_catalog(),
})
.unwrap();

View file

@ -22,7 +22,7 @@ pub use rewind::{RewindInput, RewindOutcome, rewind};
pub use source::WorkflowInput;
pub use start::{StartServices, Started, start};
pub use timeline::{ForkTarget, RunTimeline, TimelineEntry, build_timeline, timeline};
pub use validate::{ValidateInput, validate, validate_with_ready_providers};
pub use validate::{ValidateInput, validate, validate_with_catalog, validate_with_ready_providers};
pub use crate::pipeline::{LlmSpec, SandboxEnvSpec};
pub use crate::transforms::RenderMode;

View file

@ -5,12 +5,12 @@ use std::sync::Arc;
use fabro_model::{Catalog, ProviderId};
use fabro_types::WorkflowSettings;
use super::create::{preprocess_and_validate, template_context};
use super::create::{configured_default_provider, preprocess_and_validate, template_context};
use super::source::{ResolveWorkflowInput, WorkflowInput, resolve_workflow};
use crate::error::Error;
use crate::operations::RenderMode;
use crate::pipeline::Validated;
use crate::transforms::Transform;
use crate::pipeline::{TransformOptions, Validated};
use crate::transforms::{ModelResolutionTransform, Transform};
pub struct ValidateInput {
pub workflow: WorkflowInput,
@ -20,20 +20,24 @@ pub struct ValidateInput {
pub vars: HashMap<String, String>,
pub cwd: PathBuf,
pub custom_transforms: Vec<Box<dyn Transform>>,
pub catalog: Arc<Catalog>,
}
/// Parse, transform, and validate a DOT source string.
/// Parse, transform, and structurally validate a DOT source string without a
/// model catalog. Model and provider availability is left to the caller that
/// owns a catalog — typically the server.
///
/// Returns `Validated` even when validation produced errors. Call
/// `validated.raise_on_errors()` if the caller wants to fail fast.
pub fn validate(input: ValidateInput) -> Result<Validated, Error> {
let eligible_providers = input
.catalog
.all_provider_ids()
.into_iter()
.collect::<Vec<_>>();
validate_with_eligible_providers(input, &eligible_providers, false)
validate_resolving_models(input, None)
}
/// Parse, transform, and validate a DOT source string against `catalog`.
pub fn validate_with_catalog(
input: ValidateInput,
catalog: Arc<Catalog>,
) -> Result<Validated, Error> {
validate_resolving_models(input, Some(ModelResolutionTransform::new(catalog)))
}
/// Parse, transform, and validate, resolving models against the ready
@ -41,45 +45,60 @@ pub fn validate(input: ValidateInput) -> Result<Validated, Error> {
/// provider-readiness selection failures.
pub fn validate_with_ready_providers(
input: ValidateInput,
catalog: Arc<Catalog>,
ready_providers: &[ProviderId],
) -> Result<Validated, Error> {
validate_with_eligible_providers(input, ready_providers, true)
validate_resolving_models(
input,
Some(
ModelResolutionTransform::for_eligible(
catalog,
ready_providers.iter().cloned().collect(),
)
.with_catalog_fallback(true),
),
)
}
fn validate_with_eligible_providers(
/// The workflow's own default provider is only known once the workflow is
/// resolved, so callers hand in a partially built transform and it is
/// completed here.
fn validate_resolving_models(
input: ValidateInput,
eligible_providers: &[ProviderId],
catalog_fallback: bool,
model_resolution: Option<ModelResolutionTransform>,
) -> Result<Validated, Error> {
let ValidateInput {
workflow,
settings,
vars,
cwd,
custom_transforms,
} = input;
let resolved = resolve_workflow(ResolveWorkflowInput {
workflow: input.workflow,
settings: input.settings,
cwd: input.cwd,
workflow,
settings,
cwd,
})
.map_err(|err| Error::Parse(err.to_string()))?;
let model_resolution = model_resolution.map(|resolution| {
resolution.with_default_provider(configured_default_provider(&resolved.settings))
});
preprocess_and_validate(
&resolved.raw_source,
resolved
.dot_path
.as_ref()
.map(|path| path.display().to_string()),
resolved.current_dir,
resolved.file_resolver,
input.custom_transforms,
template_context(Some(&resolved.settings), input.vars),
resolved.goal_override.as_deref(),
RenderMode::Structural,
resolved
.settings
.run
.model
.provider
.as_deref()
.filter(|provider| !provider.is_empty())
.map(fabro_model::ProviderId::new),
eligible_providers,
catalog_fallback,
&input.catalog,
&TransformOptions {
current_dir: resolved.current_dir,
file_resolver: resolved.file_resolver,
template_context: template_context(Some(&resolved.settings), vars),
source_name: resolved
.dot_path
.as_ref()
.map(|path| path.display().to_string()),
render_mode: RenderMode::Structural,
custom_transforms,
model_resolution,
},
)
}

View file

@ -3,8 +3,8 @@ use std::sync::Arc;
use super::types::{Parsed, TransformOptions, Transformed};
use crate::error::Error;
use crate::transforms::{
FileInliningTransform, ImportTransform, ModelResolutionTransform,
StylesheetApplicationTransform, TemplateTransform, Transform,
FileInliningTransform, ImportTransform, StylesheetApplicationTransform, TemplateTransform,
Transform,
};
/// TRANSFORM phase: apply built-in and custom transforms to a parsed graph.
@ -63,13 +63,10 @@ pub fn transform(parsed: Parsed, options: &TransformOptions) -> Result<Transform
.apply_with_diagnostics(graph)?;
diagnostics.extend(transform_diagnostics);
let graph = StylesheetApplicationTransform.apply(graph)?;
let graph = ModelResolutionTransform::for_eligible(
Arc::clone(&options.catalog),
options.eligible_providers.clone(),
)
.with_default_provider(options.default_provider.clone())
.with_catalog_fallback(options.catalog_fallback)
.apply(graph)?;
let graph = match &options.model_resolution {
Some(model_resolution) => model_resolution.apply(graph)?,
None => graph,
};
// Custom transforms
let graph = options
@ -98,6 +95,7 @@ mod tests {
use crate::file_resolver::FilesystemFileResolver;
use crate::pipeline::parse::parse;
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
use crate::transforms::ModelResolutionTransform;
fn write_file(path: &Path, contents: &str) {
if let Some(parent) = path.parent() {
@ -112,16 +110,13 @@ mod tests {
fn transform_options() -> TransformOptions {
TransformOptions {
current_dir: None,
file_resolver: None,
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
catalog: test_catalog(),
default_provider: None,
eligible_providers: Catalog::builtin().all_provider_ids(),
catalog_fallback: false,
current_dir: None,
file_resolver: None,
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
}
}
@ -177,16 +172,9 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
catalog: test_catalog(),
default_provider: None,
eligible_providers: Catalog::builtin().all_provider_ids(),
catalog_fallback: false,
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
..transform_options()
})
.unwrap();
@ -227,21 +215,15 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from(
[(
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from([
(
"task".to_string(),
toml::Value::String("Launch".to_string()),
)],
)),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
catalog: test_catalog(),
default_provider: None,
eligible_providers: Catalog::builtin().all_provider_ids(),
catalog_fallback: false,
),
])),
..transform_options()
})
.unwrap();
@ -349,6 +331,33 @@ mod tests {
);
}
#[test]
fn structural_transform_preserves_catalog_owned_model_selection() {
let dot = r#"digraph Test {
graph [goal="Test"]
start [shape=Mdiamond]
work [prompt="Do work", model="private-model", provider="server-only"]
exit [shape=Msquare]
start -> work -> exit
}"#;
let parsed = parse(dot).unwrap();
let transformed = transform(parsed, &TransformOptions {
model_resolution: None,
..transform_options()
})
.unwrap();
let work = &transformed.graph.nodes["work"];
assert_eq!(
work.attrs.get("model").and_then(AttrValue::as_str),
Some("private-model")
);
assert_eq!(
work.attrs.get("provider").and_then(AttrValue::as_str),
Some("server-only")
);
}
#[test]
fn transform_reports_goal_self_reference_once_across_passes() {
// FileInlining renders the goal for prompt context, but TemplateTransform
@ -365,16 +374,10 @@ mod tests {
)
.unwrap();
let transformed = transform(parsed, &TransformOptions {
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Structural,
custom_transforms: vec![],
catalog: test_catalog(),
default_provider: None,
eligible_providers: Catalog::builtin().all_provider_ids(),
catalog_fallback: false,
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
render_mode: crate::operations::RenderMode::Structural,
..transform_options()
})
.unwrap();

View file

@ -1,4 +1,4 @@
use std::collections::{HashMap, HashSet};
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::Arc;
@ -28,7 +28,7 @@ use crate::runtime_store::RunStoreHandle;
use crate::services::{EngineServices, FabroRunToolServices, RunServices};
use crate::stage_execution::StageExecutionSeed;
use crate::steering_hub::SteeringHub;
use crate::transforms::{RenderMode, Transform};
use crate::transforms::{ModelResolutionTransform, RenderMode, Transform};
use crate::workflow_bundle::WorkflowBundle;
/// Output of the PARSE phase.
@ -359,18 +359,15 @@ pub struct Finalized {
/// Options for the TRANSFORM phase.
pub struct TransformOptions {
pub current_dir: Option<PathBuf>,
pub file_resolver: Option<Arc<dyn FileResolver>>,
pub template_context: TemplateContext,
pub source_name: Option<String>,
pub render_mode: RenderMode,
pub custom_transforms: Vec<Box<dyn Transform>>,
pub catalog: Arc<fabro_model::Catalog>,
pub default_provider: Option<ProviderId>,
pub eligible_providers: HashSet<ProviderId>,
/// Fall back to the full catalog when the eligible providers cannot
/// supply a requested model, instead of erroring.
pub catalog_fallback: bool,
pub current_dir: Option<PathBuf>,
pub file_resolver: Option<Arc<dyn FileResolver>>,
pub template_context: TemplateContext,
pub source_name: Option<String>,
pub render_mode: RenderMode,
pub custom_transforms: Vec<Box<dyn Transform>>,
/// Catalog-backed model resolution to perform. `None` preserves authored
/// model and provider selectors for catalog-free structural validation.
pub model_resolution: Option<ModelResolutionTransform>,
}
/// Options for the FINALIZE phase.

View file

@ -5,11 +5,15 @@ use super::types::{Transformed, Validated};
/// VALIDATE phase: run lint rules against the transformed graph.
///
/// Catalog-backed rules (model and provider availability) run only when
/// `catalog` is `Some`. Offline callers pass `None` so a workflow naming a
/// server-owned model is left for the server to judge.
///
/// **Infallible.** Always returns `Validated` with diagnostics. Caller decides
/// whether to fail via `validated.raise_on_errors()`.
pub fn validate(
transformed: Transformed,
catalog: &Catalog,
catalog: Option<&Catalog>,
extra_rules: &[&dyn LintRule],
) -> Validated {
let Transformed {
@ -17,44 +21,33 @@ pub fn validate(
source,
mut diagnostics,
} = transformed;
diagnostics.extend(fabro_validate::validate_with_catalog(
&graph,
catalog,
extra_rules,
));
diagnostics.extend(match catalog {
Some(catalog) => fabro_validate::validate_with_catalog(&graph, catalog, extra_rules),
None => fabro_validate::validate(&graph, extra_rules),
});
Validated::new(graph, source, diagnostics)
}
#[cfg(test)]
mod tests {
use fabro_model::Catalog;
use super::*;
use crate::pipeline::parse::parse;
use crate::pipeline::transform;
use crate::pipeline::types::TransformOptions;
fn test_catalog() -> std::sync::Arc<Catalog> {
std::sync::Arc::new(Catalog::from_builtin().unwrap())
}
fn run_pipeline(dot: &str) -> Validated {
let catalog = test_catalog();
let parsed = parse(dot).unwrap();
let transformed = transform::transform(parsed, &TransformOptions {
current_dir: None,
file_resolver: None,
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
catalog: std::sync::Arc::clone(&catalog),
default_provider: None,
eligible_providers: catalog.all_provider_ids(),
catalog_fallback: false,
current_dir: None,
file_resolver: None,
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: crate::operations::RenderMode::Strict,
custom_transforms: vec![],
model_resolution: None,
})
.unwrap();
validate(transformed, catalog.as_ref(), &[])
validate(transformed, None, &[])
}
#[test]

View file

@ -339,18 +339,18 @@ impl ImportTransform {
.edges
.retain(|edge| edge.from != placeholder_id && edge.to != placeholder_id);
for (node_id, node) in imported_graph.nodes {
for (node_id, mut merged_node) in imported_graph.nodes {
if node_id == start_id || node_id == exit_id {
continue;
}
let prefixed_id = format!("{placeholder_id}.{node_id}");
let mut merged_node = Node::new(&prefixed_id);
merged_node.id.clone_from(&prefixed_id);
let imported_attrs = std::mem::take(&mut merged_node.attrs);
merged_node.attrs.clone_from(&placeholder.default_attrs);
merged_node.attrs.extend(node.attrs);
merged_node.attrs.extend(imported_attrs);
Self::remap_retry_target(&mut merged_node.attrs, placeholder_id);
merged_node.classes = node.classes;
for class_name in &placeholder.class_names {
Self::push_class(&mut merged_node.classes, class_name);
}
@ -649,7 +649,10 @@ impl PreparedImport {
self.graph.nodes.iter().all(|(node_id, node)| {
ImportTransform::is_start_sentinel(node_id, node)
|| ImportTransform::is_exit_sentinel(node_id, node)
})
}) && matches!(
self.graph.edges.as_slice(),
[edge] if edge.from == self.start_id && edge.to == self.exit_id
)
}
}
@ -909,6 +912,7 @@ mod tests {
assert!(!graph.nodes.contains_key("validate"));
assert!(graph.nodes.contains_key("validate.lint"));
assert!(graph.nodes.contains_key("validate.test"));
assert_eq!(graph.nodes["validate.lint"].id, "validate.lint");
assert!(!graph.nodes.contains_key("validate.start"));
assert!(!graph.nodes.contains_key("validate.exit"));
@ -951,6 +955,72 @@ mod tests {
);
}
#[test]
fn edge_only_node_in_imported_fragment_stays_missing() {
let dir = tempfile::tempdir().unwrap();
write_file(
&dir.path().join("validate.fabro"),
r#"digraph validate {
start [shape=Mdiamond]
lint [prompt="Run clippy"]
exit [shape=Msquare]
start -> lint -> typo -> exit
}"#,
);
let graph = apply_import(
r#"digraph Deploy {
start [shape=Mdiamond]
validate [import="./validate.fabro"]
exit [shape=Msquare]
start -> validate -> exit
}"#,
dir.path(),
None,
);
assert!(!graph.nodes.contains_key("validate.typo"));
assert!(graph.nodes.contains_key("validate.lint"));
}
#[test]
fn edge_only_body_is_not_treated_as_empty_import() {
let dir = tempfile::tempdir().unwrap();
write_file(
&dir.path().join("validate.fabro"),
r"digraph validate {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> typo -> exit
}",
);
let graph = apply_import(
r#"digraph Deploy {
start [shape=Mdiamond]
validate [import="./validate.fabro"]
exit [shape=Msquare]
start -> validate -> exit
}"#,
dir.path(),
None,
);
assert!(!graph.nodes.contains_key("validate.typo"));
assert!(
graph
.edges
.iter()
.any(|edge| edge.from == "start" && edge.to == "validate.typo")
);
assert!(
graph
.edges
.iter()
.any(|edge| edge.from == "validate.typo" && edge.to == "exit")
);
}
#[test]
fn import_reports_structural_diagnostic_for_imported_prompt_templates() {
let dir = tempfile::tempdir().unwrap();
@ -1130,6 +1200,48 @@ mod tests {
);
}
#[test]
fn imported_start_and_exit_sentinels_must_be_declared() {
let dir = tempfile::tempdir().unwrap();
let host = r#"digraph Deploy {
start [shape=Mdiamond]
validate [import="./validate.fabro"]
exit [shape=Msquare]
start -> validate -> exit
}"#;
let cases = [
(
r#"digraph validate {
work [prompt="Run checks"]
exit [shape=Msquare]
start -> work -> exit
}"#,
"imported workflow must have exactly one start node, found 0",
),
(
r#"digraph validate {
start [shape=Mdiamond]
work [prompt="Run checks"]
start -> work -> exit
}"#,
"imported workflow must have exactly one exit node, found 0",
),
];
for (source, expected_error) in cases {
write_file(&dir.path().join("validate.fabro"), source);
let graph = apply_import(host, dir.path(), None);
assert_eq!(
graph.nodes["validate"]
.attrs
.get("import_error")
.and_then(AttrValue::as_str),
Some(expected_error)
);
}
}
#[test]
fn multiple_entry_nodes_poison_placeholder() {
let dir = tempfile::tempdir().unwrap();

View file

@ -52,6 +52,13 @@ impl ModelResolutionTransform {
self
}
/// The catalog this transform resolves against, so callers can run the
/// matching catalog-backed lint rules.
#[must_use]
pub fn catalog(&self) -> &Catalog {
&self.catalog
}
fn resolve_model(
&self,
model: &str,

View file

@ -19,17 +19,18 @@ use crate::static_reference::{
/// How the template-expansion pass should treat undefined input variables.
///
/// Validate is structural — it should not fail just because the user has not
/// bound `{{ inputs.* }}` yet. Run-start is strict — missing inputs are real
/// errors. Splitting the two lets validate work on a bare `.fabro` while
/// run-start preserves its current hard-fail behavior.
/// Both validate and run-create render structurally so they can report every
/// unbound `{{ inputs.* }}` variable in one pass rather than aborting on the
/// first. Run-create then promotes the resulting warnings to errors, which
/// keeps its hard-fail behavior.
#[derive(Clone, Copy, Debug)]
pub enum RenderMode {
/// Undefined inputs are hard errors. Used by run-create.
/// Undefined inputs abort the pass with a hard error. No production caller
/// uses this today; run-create promotes structural warnings instead.
Strict,
/// Undefined inputs render as empty and become warning diagnostics on the
/// returned `Validated`, so structural lints still run. Used by
/// `fabro validate`.
/// `fabro validate` and by run-create.
Structural,
}

View file

@ -388,6 +388,9 @@ fn parse_and_validate_human_gate() {
type="human"
]
ship_it [prompt="Ship the change"]
fixes [prompt="Apply the requested fixes"]
start -> review_gate
review_gate -> ship_it [label="[A] Approve"]
review_gate -> fixes [label="[F] Fix"]
@ -4854,6 +4857,7 @@ async fn manager_loop_child_workflow_e2e() {
#[tokio::test]
async fn import_e2e_through_engine() {
use fabro_workflow::pipeline::{TransformOptions, transform, validate};
use fabro_workflow::transforms::ModelResolutionTransform;
let dir = tempfile::tempdir().unwrap();
let catalog = std::sync::Arc::new(
@ -4897,21 +4901,20 @@ async fn import_e2e_through_engine() {
)
.expect("parse should succeed");
let transformed = transform(parsed, &TransformOptions {
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(std::sync::Arc::new(
current_dir: Some(dir.path().to_path_buf()),
file_resolver: Some(std::sync::Arc::new(
fabro_workflow::file_resolver::FilesystemFileResolver::new(None),
)),
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: fabro_workflow::operations::RenderMode::Strict,
custom_transforms: vec![],
catalog: std::sync::Arc::clone(&catalog),
default_provider: None,
eligible_providers: catalog.all_provider_ids(),
catalog_fallback: false,
template_context: fabro_template::TemplateContext::new(),
source_name: None,
render_mode: fabro_workflow::operations::RenderMode::Strict,
custom_transforms: vec![],
model_resolution: Some(ModelResolutionTransform::new(std::sync::Arc::clone(
&catalog,
))),
})
.unwrap();
let validated = validate(transformed, catalog.as_ref(), &[]);
let validated = validate(transformed, Some(catalog.as_ref()), &[]);
validated
.raise_on_errors()
.expect("validation should pass after imports expand");

View file

@ -361,6 +361,11 @@ fn main() {
"fabro_types::StageInferenceProjection",
&[],
),
(
"StageToolBatchProjection",
"fabro_types::StageToolBatchProjection",
&[],
),
("LlmOutputKind", "fabro_types::LlmOutputKind", &[]),
("PermissionLevel", "fabro_types::PermissionLevel", &[]),
(

View file

@ -69,10 +69,10 @@ pub mod types {
StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
StageContextWindowStaleness, StageContextWindowUnavailableReason,
StageContextWindowWarning, StageHandler, StageId, StageInferenceProjection,
StageModelUsage, StageOutcome, StageProjection, StageState, SubAgentProjection,
SubAgentStatus, SystemActorKind, SystemIntegrationStatus, SystemIntegrationsResponse,
TodoListProjection, TurnId, UpdateVariableRequest, UserPrincipal, Variable,
VariableListResponse, WorkflowSettings,
StageModelUsage, StageOutcome, StageProjection, StageState, StageToolBatchProjection,
SubAgentProjection, SubAgentStatus, SystemActorKind, SystemIntegrationStatus,
SystemIntegrationsResponse, TodoListProjection, TurnId, UpdateVariableRequest,
UserPrincipal, Variable, VariableListResponse, WorkflowSettings,
};
pub use crate::generated::types::*;

View file

@ -18,6 +18,7 @@ use fabro_api::types::{
StageContextWindowUnavailableReason as ApiStageContextWindowUnavailableReason,
StageContextWindowWarning as ApiStageContextWindowWarning,
StageInferenceProjection as ApiStageInferenceProjection, StageProjection as ApiStageProjection,
StageToolBatchProjection as ApiStageToolBatchProjection,
SubAgentProjection as ApiSubAgentProjection, SubAgentStatus as ApiSubAgentStatus,
TodoListProjection as ApiTodoListProjection,
};
@ -29,8 +30,8 @@ use fabro_types::{
ParallelBranchResult, PermissionLevel, SkillsProjection, StageContextWindow,
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
StageContextWindowWarning, StageInferenceProjection, StageProjection, SubAgentProjection,
SubAgentStatus, TodoListKind, TodoListProjection,
StageContextWindowWarning, StageInferenceProjection, StageProjection, StageToolBatchProjection,
SubAgentProjection, SubAgentStatus, TodoListKind, TodoListProjection,
};
use serde_json::json;
@ -68,9 +69,32 @@ fn stage_projection_reuses_nested_agent_state_types() {
);
assert_same_type::<ApiStageContextWindowWarning, StageContextWindowWarning>();
assert_same_type::<ApiStageInferenceProjection, StageInferenceProjection>();
assert_same_type::<ApiStageToolBatchProjection, StageToolBatchProjection>();
assert_same_type::<ApiLlmOutputKind, LlmOutputKind>();
}
#[test]
fn stage_tool_batch_projection_matches_openapi_json_shape() {
let batch = StageToolBatchProjection {
session_id: "ses_root".to_string(),
started_at: "2026-04-29T12:34:00Z".parse().unwrap(),
open_call_ids: ["call_1".to_string(), "call_2".to_string()]
.into_iter()
.collect(),
};
let value = serde_json::to_value(&batch).unwrap();
assert_eq!(
value,
json!({
"session_id": "ses_root",
"started_at": "2026-04-29T12:34:00Z",
"open_call_ids": ["call_1", "call_2"]
})
);
let api_batch: ApiStageToolBatchProjection = serde_json::from_value(value).unwrap();
assert_eq!(api_batch, batch);
}
#[test]
fn stage_inference_projection_matches_openapi_json_shape() {
let inference = StageInferenceProjection {
@ -148,6 +172,10 @@ fn stage_projection_without_inference_round_trips() {
let stage: StageProjection = serde_json::from_value(value.clone()).unwrap();
assert!(stage.inference.is_none());
assert!(stage.acp_started_at.is_none());
assert!(stage.tool_batch.is_none());
assert_eq!(stage.live_inference_ms, 0);
assert_eq!(stage.live_tool_ms, 0);
assert_eq!(serde_json::to_value(stage).unwrap(), value);
}
@ -311,6 +339,7 @@ fn stage_projection_round_trips_representative_json() {
"first_output_kind": "text",
"retries": 0
},
"acp_started_at": "2026-04-29T12:34:00Z",
"agent_control": "running",
"state": "succeeded"
});

View file

@ -125,7 +125,7 @@ pub use run_projection::{
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
StageContextWindowWarning, StageInferenceProjection, StageModelUsage, StageProjection,
SubAgentProjection, SubAgentStatus, first_event_seq,
StageToolBatchProjection, SubAgentProjection, SubAgentStatus, first_event_seq,
};
pub use run_sandbox::{
RunSandbox, RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan,

View file

@ -926,8 +926,6 @@ impl<'de> Deserialize<'de> for RunEvent {
#[cfg(test)]
mod tests {
use std::collections::HashMap;
use serde_json::json;
use super::*;
@ -992,20 +990,9 @@ mod tests {
#[test]
fn run_event_deserializes_adjacent_layout() {
let settings = WorkflowSettings::default();
let graph = Graph {
name: "test".to_string(),
nodes: HashMap::from([("start".to_string(), Node {
id: "start".to_string(),
attrs: HashMap::new(),
classes: Vec::new(),
})]),
edges: vec![Edge {
from: "start".to_string(),
to: "done".to_string(),
attrs: HashMap::new(),
}],
attrs: HashMap::new(),
};
let mut graph = Graph::new("test");
graph.nodes.insert("start".to_string(), Node::new("start"));
graph.edges.push(Edge::new("start", "done"));
let line = json!({
"id": "evt_2",

View file

@ -1,9 +1,9 @@
use std::borrow::Cow;
use std::collections::{BTreeMap, HashMap};
use std::collections::{BTreeMap, BTreeSet, HashMap};
use std::num::NonZeroU32;
use chrono::{DateTime, Utc};
use fabro_model::{ReasoningEffort, Speed};
use fabro_model::{Catalog, ReasoningEffort, Speed};
use strum::{Display, EnumString, IntoStaticStr};
use crate::run_event::{AgentSessionActivatedProps, StagePromptProps};
@ -12,7 +12,7 @@ use crate::{
AgentToolSummary, BilledTokenCounts, Checkpoint, Conclusion, InterviewQuestionRecord,
InvalidTransition, LlmOutputKind, ModelRef, PermissionLevel, PullRequestLink, RunApproval,
RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, RunTiming, StageCompletion,
StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection,
StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection, timing,
};
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
@ -354,10 +354,31 @@ pub struct StageProjection {
/// immutable projections under their own `StageId`s.
///
/// `None` for stages still in flight (`started_at` is set but no terminal
/// event has been observed yet). For live wall-time ticking, the UI uses
/// `started_at`; once terminal this carries the finalized breakdown.
/// event has been observed yet). For a live breakdown while in flight, use
/// [`StageProjection::live_timing`]; once terminal this carries the
/// finalized, authoritative breakdown.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub timing: Option<StageTiming>,
/// Inference time accumulated from closed brackets during this attempt.
///
/// Live estimate only: the authoritative value arrives with the terminal
/// event and lands in `timing`. Excludes the currently-open bracket, which
/// [`StageProjection::live_timing`] adds from `inference.started_at`.
#[serde(default, skip_serializing_if = "is_zero_ms")]
pub live_inference_ms: u64,
/// Tool time accumulated from closed tool batches during this attempt.
///
/// A batch spans the first `agent.tool.started` with no outstanding calls
/// through the `agent.tool.completed` that drains the last one, so tools
/// running concurrently within a turn are counted once. This matches how
/// the in-process stopwatch brackets `execute_tool_calls`; summing
/// per-call durations would over-count parallel tool use.
#[serde(default, skip_serializing_if = "is_zero_ms")]
pub live_tool_ms: u64,
/// Open tool batch for this stage: when the current batch started, and the
/// `tool_call_id`s that have not yet reported completion.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub tool_batch: Option<StageToolBatchProjection>,
#[serde(default)]
pub usage: BilledTokenCounts,
#[serde(default, skip_serializing_if = "Option::is_none")]
@ -390,11 +411,45 @@ pub struct StageProjection {
/// the authority on whether a run is actually stuck.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub inference: Option<StageInferenceProjection>,
/// Start of an external ACP agent process, if one is running.
///
/// ACP agents do not emit Fabro's internal LLM brackets, so their process
/// lifetime is the best available live inference estimate.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub acp_started_at: Option<DateTime<Utc>>,
#[serde(default)]
pub agent_control: AgentControlState,
pub state: StageState,
}
/// Serde guard so zero-valued live accumulators stay off the wire.
#[allow(
clippy::trivially_copy_pass_by_ref,
reason = "serde skip_serializing_if predicates receive fields by reference"
)]
fn is_zero_ms(value: &u64) -> bool {
*value == 0
}
/// One open tool batch: tool calls dispatched together that have not all
/// reported completion.
///
/// `open_call_ids` is a set rather than a count because `agent.tool.completed`
/// identifies its call by id, and a projection replaying a truncated or
/// duplicated log must not let a repeated completion drain the batch early.
#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
pub struct StageToolBatchProjection {
/// Root agent session that dispatched the batch. Later transitions are
/// gated on it so delayed events from a replaced session cannot mutate
/// the current session's batch.
pub session_id: String,
/// When the batch opened — the first `agent.tool.started` observed while
/// no other calls were outstanding.
pub started_at: DateTime<Utc>,
/// Calls dispatched but not yet completed, by `tool_call_id`.
pub open_call_ids: BTreeSet<String>,
}
/// One open inference bracket: a dispatched LLM request that has not yet
/// produced a message, error, or interrupt.
#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)]
@ -506,6 +561,9 @@ impl StageProjection {
response: None,
completion: None,
timing: None,
live_inference_ms: 0,
live_tool_ms: 0,
tool_batch: None,
usage: BilledTokenCounts::default(),
model: None,
root_agent_todos: None,
@ -516,6 +574,7 @@ impl StageProjection {
mcp_servers: Vec::new(),
context_window: None,
inference: None,
acp_started_at: None,
agent_control: AgentControlState::default(),
provider_used: None,
diff: None,
@ -540,6 +599,29 @@ impl StageProjection {
self.state
}
/// This stage's token counts with a cost attached.
///
/// A provider-reported cost always wins. Otherwise the catalog prices the
/// recorded tokens for the stage's model. The stored counts pass through
/// untouched when there is no catalog, no model, or no price for that
/// model. Empty usage also passes through untouched. These cases leave
/// `total_usd_micros` as `None` rather than zero.
#[must_use]
pub fn billed_usage(&self, catalog: Option<&Catalog>) -> Cow<'_, BilledTokenCounts> {
if self.usage.total_usd_micros.is_some() || self.usage.is_zero() {
return Cow::Borrowed(&self.usage);
}
let (Some(catalog), Some(model)) = (catalog, self.model.as_ref()) else {
return Cow::Borrowed(&self.usage);
};
let Some(total_usd_micros) = catalog.price_tokens(model, &self.usage.token_counts()) else {
return Cow::Borrowed(&self.usage);
};
let mut usage = self.usage.clone();
usage.total_usd_micros = Some(total_usd_micros);
Cow::Owned(usage)
}
/// Live wall-clock time in milliseconds.
///
/// While the stage is non-terminal (`Pending`, `Running`, or `Retrying`),
@ -555,14 +637,197 @@ impl StageProjection {
state,
StageState::Running | StageState::Retrying | StageState::Pending
) {
return self.started_at.map(|started| {
u64::try_from(now.signed_duration_since(started).num_milliseconds().max(0))
.unwrap_or(0)
});
return self
.started_at
.map(|started| timing::elapsed_ms(started, now));
}
self.timing.map(|timing| timing.wall_time_ms)
}
/// Live timing breakdown in milliseconds — the active-time twin of
/// [`Self::live_wall_time_ms`].
///
/// Once terminal, returns the stored `timing` unchanged: the finalized
/// breakdown comes from the worker's own stopwatch and is authoritative.
///
/// While in flight, returns an estimate reconstructed from the event log:
/// accumulated closed brackets plus whatever bracket is open right now.
/// The estimate is per-handler, because only agent stages emit brackets at
/// all:
///
/// - `Agent` — accumulated inference and tool brackets, plus the open
/// inference bracket and open tool batch.
/// - `Prompt` — one inference call spanning the stage, so elapsed time
/// since `started_at` counts as inference. Matches the finalized
/// `active_only(inference, 0)`.
/// - `Command` — the command *is* the work, so elapsed time counts as tool.
/// Matches the finalized `active_only(0, duration_ms)`.
/// - Everything else — zero. Waiting on a human, a timer, a condition, or
/// child branches is wall time, not active time.
///
/// Active is clamped to wall. A worker killed mid-turn leaves its bracket
/// open forever (see [`StageInferenceProjection`]), and without the clamp
/// that bracket would tick up without bound. The clamp does not need to
/// know the worker died: a stage cannot have been active longer than it
/// has existed. `watchdog.timeout` remains the authority on whether a run
/// is actually stuck.
#[must_use]
pub fn live_timing(&self, now: DateTime<Utc>) -> StageTiming {
if let Some(timing) = self.timing {
return timing;
}
let wall_time_ms = self.live_wall_time_ms(now).unwrap_or(0);
// `handler` is absent on projections built from events written before
// stage execution identity. Treat those as agent stages, matching
// `StageHandler::from_handler_type`: the accumulators below are only
// ever populated by agent events, so a legacy non-agent stage still
// reads zero rather than being credited work it never did.
let handler = self.handler.unwrap_or(StageHandler::Agent);
let (inference_time_ms, tool_time_ms) = match handler {
StageHandler::Agent => {
let open_inference = self
.inference
.as_ref()
.map(|inference| inference.started_at)
.or(self.acp_started_at)
.map_or(0, |started_at| timing::elapsed_ms(started_at, now));
let open_tool = self
.tool_batch
.as_ref()
.map_or(0, |batch| timing::elapsed_ms(batch.started_at, now));
(
self.live_inference_ms.saturating_add(open_inference),
self.live_tool_ms.saturating_add(open_tool),
)
}
StageHandler::Prompt => (wall_time_ms, 0),
StageHandler::Command => (0, wall_time_ms),
// Waiting on a human, a timer, a condition, or child branches is
// wall time, not active time.
StageHandler::Human
| StageHandler::Wait
| StageHandler::Conditional
| StageHandler::Parallel
| StageHandler::ParallelFanIn
| StageHandler::StackManagerLoop
| StageHandler::Start
| StageHandler::Exit => (0, 0),
};
StageTiming::new(wall_time_ms, inference_time_ms, tool_time_ms).clamped_to_wall()
}
/// Fold a closed inference bracket into the live accumulator.
pub fn accumulate_inference_ms(&mut self, elapsed_ms: u64) {
self.live_inference_ms = self.live_inference_ms.saturating_add(elapsed_ms);
}
/// Record the start of an external ACP agent process.
pub fn open_acp_inference(&mut self, started_at: DateTime<Utc>) {
self.close_open_acp_inference(started_at);
self.acp_started_at = Some(started_at);
}
/// Close an ACP process with its measured duration.
///
/// The duration covers the complete process. Use the larger value so a
/// replayed terminal event or a projection restored mid-process cannot
/// double-count it.
pub fn close_acp_inference(&mut self, duration_ms: u64) {
self.acp_started_at = None;
self.live_inference_ms = self.live_inference_ms.max(duration_ms);
}
/// Fold an open ACP process into the live accumulator at a run boundary.
pub fn close_open_acp_inference(&mut self, now: DateTime<Utc>) {
let Some(started_at) = self.acp_started_at.take() else {
return;
};
self.accumulate_inference_ms(timing::elapsed_ms(started_at, now));
}
/// Record a dispatched tool call, opening a batch if none is outstanding.
///
/// If a replacement root session starts work before the old session's end
/// event arrives, freeze the old batch at this boundary before opening
/// the new one. This keeps the sessions separate without dropping time.
pub fn open_tool_call(
&mut self,
session_id: String,
tool_call_id: String,
started_at: DateTime<Utc>,
) {
let replaces_open_batch = self
.tool_batch
.as_ref()
.is_some_and(|batch| batch.session_id != session_id);
if replaces_open_batch {
self.close_open_tool_batch(started_at);
}
self.tool_batch
.get_or_insert_with(|| StageToolBatchProjection {
session_id,
started_at,
open_call_ids: BTreeSet::new(),
})
.open_call_ids
.insert(tool_call_id);
}
/// Retire a tool call. Folds the batch into the live accumulator once the
/// last outstanding call reports, so concurrent calls count once.
pub fn close_tool_call(&mut self, session_id: &str, tool_call_id: &str, now: DateTime<Utc>) {
let Some(batch) = self.tool_batch.as_mut() else {
return;
};
if batch.session_id != session_id
|| !batch.open_call_ids.remove(tool_call_id)
|| !batch.open_call_ids.is_empty()
{
return;
}
self.close_open_tool_batch(now);
}
/// Close a tool batch only when it belongs to `session_id`.
pub fn close_tool_batch_for_session(&mut self, session_id: &str, now: DateTime<Utc>) {
let opened_here = self
.tool_batch
.as_ref()
.is_some_and(|batch| batch.session_id == session_id);
if opened_here {
self.close_open_tool_batch(now);
}
}
/// Fold any open tool batch into the live accumulator.
pub fn close_open_tool_batch(&mut self, now: DateTime<Utc>) {
let Some(batch) = self.tool_batch.take() else {
return;
};
self.live_tool_ms = self
.live_tool_ms
.saturating_add(timing::elapsed_ms(batch.started_at, now));
}
/// Install a worker-provided terminal timing and discard transient live
/// bookkeeping that is no longer authoritative.
pub fn set_authoritative_timing(&mut self, timing: StageTiming) {
self.timing = Some(timing);
self.clear_live_timing();
}
/// Discard transient timing accumulators and open brackets.
pub fn clear_live_timing(&mut self) {
self.live_inference_ms = 0;
self.live_tool_ms = 0;
self.inference = None;
self.acp_started_at = None;
self.tool_batch = None;
}
/// Begin a new automatic attempt within this stage execution: clear every
/// per-attempt field so prior-attempt data does not leak, then record
/// `started_at` and `state = Running`. Preserves `first_event_seq`
@ -709,18 +974,21 @@ impl RunProjection {
/// terminal conclusion yet.
///
/// Run-level wall time ticks from `run.started` to `now`. Active time sums
/// inference and tool timing from stages that have already emitted a
/// terminal stage event. Stage projections do not currently track live
/// inference/tool time while a stage is still running, so active time steps
/// forward when each stage completes while wall time advances continuously.
/// [`StageProjection::live_timing`] across every stage, so an in-flight
/// stage contributes its live estimate rather than nothing — both halves
/// advance continuously. Terminal stages contribute their finalized,
/// authoritative breakdown.
///
/// Active is not clamped to run wall time here: concurrent branches can
/// legitimately sum past it. The clamp applies per stage.
#[must_use]
pub fn live_run_timing(&self, now: DateTime<Utc>) -> Option<RunTiming> {
let start = self.start.as_ref()?;
let wall_time_ms = RunTiming::wall_time_ms_since(start.start_time, now);
let wall_time_ms = timing::elapsed_ms(start.start_time, now);
let active = self
.stages
.values()
.filter_map(|stage| stage.timing)
.map(|stage| stage.live_timing(now))
.fold(RunTiming::default(), |acc, timing| {
acc.saturating_add(&RunTiming::from(timing))
});
@ -844,11 +1112,13 @@ mod iter_stages_tests {
use std::num::NonZeroU32;
use chrono::Utc;
use fabro_model::{Catalog, ModelRef, ProviderId};
use serde_json::json;
use super::RunProjection;
use crate::{
AgentControlState, Graph, RunId, RunSpec, StageProjection, WorkflowSettings, test_support,
AgentControlState, BilledTokenCounts, Graph, RunId, RunSpec, StageProjection,
WorkflowSettings, test_support,
};
fn seq(n: u32) -> NonZeroU32 {
@ -970,4 +1240,361 @@ mod iter_stages_tests {
assert_eq!(order, vec!["build@1", "verify@1", "verify@2"]);
}
}
fn priced_stage(total_usd_micros: Option<i64>) -> StageProjection {
let mut stage = StageProjection::new(seq(1));
stage.usage = BilledTokenCounts {
input_tokens: 500_000,
output_tokens: 125_000,
total_tokens: 625_000,
total_usd_micros,
..BilledTokenCounts::default()
};
stage.model = Some(ModelRef {
provider: ProviderId::openai(),
model_id: "gpt-5.4".into(),
speed: None,
});
stage
}
#[test]
fn billed_usage_prices_uncosted_tokens_from_the_catalog() {
let stage = priced_stage(None);
assert_eq!(stage.billed_usage(None).total_usd_micros, None);
let priced = stage.billed_usage(Some(Catalog::builtin()));
assert!(
priced.total_usd_micros.is_some_and(|cost| cost > 0),
"expected a catalog price, got {:?}",
priced.total_usd_micros
);
// Pricing only fills in the cost; the token buckets pass through.
assert_eq!(priced.input_tokens, 500_000);
assert_eq!(priced.output_tokens, 125_000);
}
#[test]
fn billed_usage_keeps_a_provider_reported_cost_over_the_catalog_estimate() {
let stage = priced_stage(Some(42));
assert_eq!(
stage
.billed_usage(Some(Catalog::builtin()))
.total_usd_micros,
Some(42)
);
}
#[test]
fn billed_usage_leaves_a_modelless_stage_uncosted() {
let mut stage = priced_stage(None);
stage.model = None;
assert_eq!(
stage
.billed_usage(Some(Catalog::builtin()))
.total_usd_micros,
None
);
}
#[test]
fn billed_usage_leaves_zero_tokens_uncosted() {
let mut stage = priced_stage(None);
stage.usage = BilledTokenCounts::default();
assert_eq!(
stage
.billed_usage(Some(Catalog::builtin()))
.total_usd_micros,
None
);
}
}
#[cfg(test)]
mod live_timing_tests {
use std::collections::HashMap;
use chrono::{DateTime, TimeZone, Utc};
use super::{RunProjection, StageToolBatchProjection};
use crate::{
Graph, ModelRef, RunId, RunSpec, StageHandler, StageInferenceProjection, StageProjection,
StageState, StageTiming, StartRecord, WorkflowSettings, first_event_seq, test_support,
};
fn at(seconds: i64) -> DateTime<Utc> {
Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap()
}
fn projection() -> RunProjection {
RunProjection::new(
"Test run".to_string(),
RunSpec {
run_id: RunId::new(),
settings: WorkflowSettings::default(),
graph: Graph::new("test"),
graph_source: None,
workflow_slug: None,
automation: None,
source_directory: None,
labels: HashMap::default(),
provenance: test_support::test_run_provenance(),
manifest_blob: None,
definition_blob: None,
git: None,
fork_source_ref: None,
},
at(0),
)
}
/// In-flight stage that started at `at(0)`.
fn running(handler: StageHandler) -> StageProjection {
let mut stage = StageProjection::new(first_event_seq(1));
stage.handler = Some(handler);
stage.started_at = Some(at(0));
stage.state = StageState::Running;
stage
}
fn open_bracket(started_at: DateTime<Utc>) -> StageInferenceProjection {
StageInferenceProjection {
session_id: "session-1".to_string(),
started_at,
requested_model: ModelRef {
provider: "anthropic".parse().unwrap(),
model_id: "claude-sonnet-5".into(),
speed: None,
},
first_output_at: None,
first_output_kind: None,
retries: 0,
}
}
#[test]
fn terminal_stage_returns_stored_timing_unchanged() {
let mut stage = running(StageHandler::Agent);
stage.state = StageState::Succeeded;
stage.timing = Some(StageTiming::new(90_000, 78_000, 7_000));
// Live accumulators are stale leftovers; the finalized value wins.
stage.live_inference_ms = 5;
stage.live_tool_ms = 5;
assert_eq!(
stage.live_timing(at(600)),
StageTiming::new(90_000, 78_000, 7_000)
);
}
#[test]
fn agent_stage_sums_accumulators_and_open_brackets() {
let mut stage = running(StageHandler::Agent);
stage.live_inference_ms = 30_000;
stage.live_tool_ms = 5_000;
// Inference open for 20s, tools open for 10s, at t=120s.
stage.inference = Some(open_bracket(at(100)));
stage.tool_batch = Some(StageToolBatchProjection {
session_id: "session-1".to_string(),
started_at: at(110),
open_call_ids: ["call-1".to_string()].into_iter().collect(),
});
assert_eq!(
stage.live_timing(at(120)),
StageTiming::new(120_000, 50_000, 15_000)
);
}
#[test]
fn agent_stage_without_brackets_reports_only_accumulators() {
let mut stage = running(StageHandler::Agent);
stage.live_inference_ms = 30_000;
stage.live_tool_ms = 5_000;
assert_eq!(
stage.live_timing(at(120)),
StageTiming::new(120_000, 30_000, 5_000)
);
}
#[test]
fn acp_process_counts_as_live_inference_and_uses_measured_duration() {
let mut stage = running(StageHandler::Agent);
stage.open_acp_inference(at(30));
assert_eq!(
stage.live_timing(at(45)),
StageTiming::new(45_000, 15_000, 0)
);
stage.close_acp_inference(14_500);
assert_eq!(
stage.live_timing(at(45)),
StageTiming::new(45_000, 14_500, 0)
);
}
#[test]
fn prompt_stage_counts_elapsed_as_inference() {
let stage = running(StageHandler::Prompt);
assert_eq!(
stage.live_timing(at(45)),
StageTiming::new(45_000, 45_000, 0)
);
}
#[test]
fn command_stage_counts_elapsed_as_tool() {
let stage = running(StageHandler::Command);
assert_eq!(
stage.live_timing(at(45)),
StageTiming::new(45_000, 0, 45_000)
);
}
#[test]
fn waiting_handlers_report_wall_time_with_zero_active() {
for handler in [
StageHandler::Human,
StageHandler::Wait,
StageHandler::Conditional,
StageHandler::Parallel,
StageHandler::ParallelFanIn,
StageHandler::StackManagerLoop,
StageHandler::Start,
StageHandler::Exit,
] {
let stage = running(handler);
let timing = stage.live_timing(at(600));
assert_eq!(
timing,
StageTiming::new(600_000, 0, 0),
"{handler} should report wall time only"
);
}
}
#[test]
fn open_bracket_from_a_killed_worker_is_clamped_to_wall() {
let mut stage = running(StageHandler::Agent);
// Bracket opened before the stage even started — the pathological
// shape a killed worker leaves behind. Without the clamp this would
// report 700s of inference against 600s of wall.
stage.inference = Some(open_bracket(at(-100)));
let timing = stage.live_timing(at(600));
assert_eq!(timing.wall_time_ms, 600_000);
assert_eq!(timing.active_time_ms, 600_000);
}
#[test]
fn clamping_preserves_the_inference_tool_split() {
let mut stage = running(StageHandler::Agent);
// 3:1 inference:tool, totalling 200s of active against 100s of wall.
stage.live_inference_ms = 150_000;
stage.live_tool_ms = 50_000;
let timing = stage.live_timing(at(100));
assert_eq!(timing.wall_time_ms, 100_000);
assert_eq!(timing.active_time_ms, 100_000);
assert_eq!(timing.inference_time_ms, 75_000);
assert_eq!(timing.tool_time_ms, 25_000);
}
#[test]
fn live_run_timing_counts_in_flight_stages_not_just_terminal_ones() {
// The shape that motivated this change: two finished stages and one
// long-running agent stage that had been active nearly the whole run.
let mut projection = projection();
projection.start = Some(StartRecord {
start_time: at(0),
run_branch: None,
base_sha: None,
});
let baseline = projection.stage_entry("baseline", 1, first_event_seq(1));
baseline.handler = Some(StageHandler::Command);
baseline.state = StageState::Succeeded;
baseline.timing = Some(StageTiming::new(42_666, 0, 42_663));
let assess = projection.stage_entry("assess", 1, first_event_seq(2));
assess.handler = Some(StageHandler::Agent);
assess.state = StageState::Succeeded;
assess.timing = Some(StageTiming::new(86_025, 78_230, 7_588));
let plan = projection.stage_entry("plan", 1, first_event_seq(3));
plan.handler = Some(StageHandler::Agent);
plan.started_at = Some(at(146));
plan.state = StageState::Running;
plan.live_inference_ms = 700_000;
plan.live_tool_ms = 150_000;
let timing = projection.live_run_timing(at(1_013)).unwrap();
assert_eq!(timing.wall_time_ms, 1_013_000);
assert_eq!(timing.inference_time_ms, 778_230);
assert_eq!(timing.tool_time_ms, 200_251);
// Before this change the in-flight stage contributed nothing and the
// run reported 128,481 ms of active time against 1,013,000 ms of wall.
assert_eq!(timing.active_time_ms, 978_481);
}
#[test]
fn live_run_timing_may_exceed_run_wall_when_branches_overlap() {
let mut projection = projection();
projection.start = Some(StartRecord {
start_time: at(0),
run_branch: None,
base_sha: None,
});
for (index, node) in ["branch-a", "branch-b", "branch-c"].iter().enumerate() {
let stage =
projection.stage_entry(node, 1, first_event_seq(u32::try_from(index).unwrap() + 1));
stage.handler = Some(StageHandler::Agent);
stage.state = StageState::Succeeded;
stage.timing = Some(StageTiming::new(60_000, 60_000, 0));
}
let timing = projection.live_run_timing(at(60)).unwrap();
assert_eq!(timing.wall_time_ms, 60_000);
assert_eq!(
timing.active_time_ms, 180_000,
"concurrent branches legitimately sum past run wall time"
);
}
#[test]
fn a_legacy_stage_without_a_recorded_handler_uses_its_accumulators() {
let mut stage = StageProjection::new(first_event_seq(1));
stage.handler = None;
stage.started_at = Some(at(0));
stage.state = StageState::Running;
stage.live_inference_ms = 30_000;
assert_eq!(
stage.live_timing(at(120)),
StageTiming::new(120_000, 30_000, 0)
);
}
#[test]
fn a_legacy_stage_with_no_accumulators_reports_no_active_time() {
let mut stage = StageProjection::new(first_event_seq(1));
stage.handler = None;
stage.started_at = Some(at(0));
stage.state = StageState::Running;
assert_eq!(stage.live_timing(at(120)), StageTiming::new(120_000, 0, 0));
}
}

View file

@ -18,6 +18,16 @@
use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};
/// Non-negative milliseconds between two instants.
///
/// Clock skew or an out-of-order replay can put `end` before `start`; those
/// spans contribute zero rather than wrapping into a large unsigned value.
#[must_use]
pub fn elapsed_ms(start: DateTime<Utc>, end: DateTime<Utc>) -> u64 {
u64::try_from(end.signed_duration_since(start).num_milliseconds().max(0))
.expect("non-negative chrono millisecond durations fit in u64")
}
/// Timing breakdown for one stage visit.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct StageTiming {
@ -62,6 +72,34 @@ impl StageTiming {
Self::new(0, inference_time_ms, tool_time_ms)
}
/// Scale the breakdown down so `active_time_ms` does not exceed
/// `wall_time_ms`, preserving the inference/tool ratio.
///
/// Only meaningful for live estimates of a single in-flight stage, where
/// an open bracket left behind by a killed worker would otherwise tick up
/// without bound. Finalized timings come from the worker's stopwatch and
/// already satisfy the invariant.
///
/// Deliberately *not* applied at run level: concurrent branches can
/// legitimately sum past run wall time.
#[must_use]
pub fn clamped_to_wall(&self) -> Self {
let active_time_ms = u128::from(self.inference_time_ms) + u128::from(self.tool_time_ms);
if active_time_ms <= u128::from(self.wall_time_ms) {
return *self;
}
// Preserve the split rather than truncating one side, so a clamped
// stage still shows where its time went. Widen for the multiply: the
// quotient is bounded by `wall_time_ms` because the exact, widened
// active total exceeds it here, so it always fits back into u64.
let scaled =
u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) / active_time_ms;
let inference_time_ms =
u64::try_from(scaled).expect("scaled inference time is bounded by wall time");
let tool_time_ms = self.wall_time_ms.saturating_sub(inference_time_ms);
Self::new(self.wall_time_ms, inference_time_ms, tool_time_ms)
}
/// Sum two timings field-by-field. Used to aggregate visits of one node
/// and to accumulate run-level rollups.
#[must_use]
@ -136,13 +174,6 @@ impl RunTiming {
..self
}
}
/// Milliseconds elapsed from `start` to `now`, clamped at zero.
#[must_use]
pub fn wall_time_ms_since(start: DateTime<Utc>, now: DateTime<Utc>) -> u64 {
u64::try_from(now.signed_duration_since(start).num_milliseconds().max(0))
.expect("non-negative milliseconds fit in u64")
}
}
impl From<StageTiming> for RunTiming {
@ -158,7 +189,9 @@ impl From<StageTiming> for RunTiming {
#[cfg(test)]
mod tests {
use super::{RunTiming, StageTiming};
use chrono::{TimeZone, Utc};
use super::{RunTiming, StageTiming, elapsed_ms};
#[test]
fn stage_timing_new_derives_active_as_sum_of_inference_and_tool() {
@ -189,6 +222,23 @@ mod tests {
assert_eq!(sum.active_time_ms, 175);
}
#[test]
fn stage_timing_clamp_uses_the_unsaturated_active_total() {
let timing = StageTiming::new(u64::MAX, u64::MAX, u64::MAX).clamped_to_wall();
assert_eq!(timing.inference_time_ms, u64::MAX / 2);
assert_eq!(timing.tool_time_ms, u64::MAX.saturating_sub(u64::MAX / 2));
assert_eq!(timing.active_time_ms, u64::MAX);
}
#[test]
fn elapsed_ms_clamps_out_of_order_instants_to_zero() {
let later = Utc.timestamp_opt(100, 0).unwrap();
let earlier = Utc.timestamp_opt(99, 0).unwrap();
assert_eq!(elapsed_ms(later, earlier), 0);
}
#[test]
fn run_timing_wall_only_zeroes_breakdown_and_active() {
let timing = RunTiming::wall_only(1500);

View file

@ -4,6 +4,10 @@
// blank lines at end of file. Left alone, every regeneration produces a noisy
// whitespace diff that masks real spec/client drift. This pass strips trailing
// whitespace from every line and ends each file with exactly one newline.
//
// openapi-generator maps arrays with `uniqueItems` to `Set<T>`, but Axios
// decodes and encodes their JSON representation as arrays. Keep the generated
// type aligned with the runtime wire value.
import { Glob } from "bun";
@ -14,6 +18,7 @@ for await (const path of glob.scan(".")) {
const original = await Bun.file(path).text();
const normalized =
original
.replace(/\bSet</g, "Array<")
.split("\n")
.map((line) => line.replace(/\s+$/, ""))
.join("\n")

View file

@ -487,6 +487,7 @@ models/stage-projection.ts
models/stage-state.ts
models/stage-summary.ts
models/stage-timing.ts
models/stage-tool-batch-projection.ts
models/start-record.ts
models/start-run-request.ts
models/steer-run-request.ts

View file

@ -21,7 +21,7 @@ export interface BatchDeleteRunsRequest {
/**
* Run IDs to process, in result order.
*/
'run_ids': Set<string>;
'run_ids': Array<string>;
/**
* Whether to force deletion of active runs. Defaults to `false`.
*/

View file

@ -21,5 +21,5 @@ export interface BatchRunLifecycleRequest {
/**
* Run IDs to process, in result order.
*/
'run_ids': Set<string>;
'run_ids': Array<string>;
}

View file

@ -457,6 +457,7 @@ export * from './stage-projection';
export * from './stage-state';
export * from './stage-summary';
export * from './stage-timing';
export * from './stage-tool-batch-projection';
export * from './start-record';
export * from './start-run-request';
export * from './steer-run-request';

View file

@ -13,6 +13,9 @@
*/
// May contain unused imports in some cases
// @ts-ignore
import type { BilledTokenCounts } from './billed-token-counts';
// May contain unused imports in some cases
// @ts-ignore
import type { StageHandler } from './stage-handler';
@ -62,4 +65,8 @@ export interface RunStage {
* Wall-clock time the latest attempt of this stage started, if known.
*/
'started_at'?: string | null;
/**
* Token counts for this stage execution alone. `total_usd_micros` is the provider-reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/billing` rows use. All-zero counts mean the stage made no model calls. Unlike the billing rows, which sum every visit of a node, this covers only this visit.
*/
'billing': BilledTokenCounts;
}

View file

@ -15,12 +15,12 @@
/**
* Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently.
* Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. For a running run, stages still in flight contribute a live estimate rather than nothing, so wall and active both advance continuously. Unlike `StageTiming`, active is not clamped to wall here — concurrent branches can legitimately sum past run wall time.
*/
export interface RunTiming {
'wall_time_ms': number;
'inference_time_ms'?: number;
'tool_time_ms'?: number;
'inference_time_ms': number;
'tool_time_ms': number;
/**
* Equals `inference_time_ms + tool_time_ms`.
*/

View file

@ -60,6 +60,9 @@ import type { StageState } from './stage-state';
import type { StageTiming } from './stage-timing';
// May contain unused imports in some cases
// @ts-ignore
import type { StageToolBatchProjection } from './stage-tool-batch-projection';
// May contain unused imports in some cases
// @ts-ignore
import type { SubAgentProjection } from './sub-agent-projection';
// May contain unused imports in some cases
// @ts-ignore
@ -96,6 +99,15 @@ export interface StageProjection {
*/
'started_at'?: string | null;
'timing'?: StageTiming | null;
/**
* Inference time accumulated from closed brackets during the current attempt. Live estimate only — the authoritative value arrives with the terminal event and lands in `timing`. Excludes the currently open bracket, whose span is measured from `inference.started_at`.
*/
'live_inference_ms'?: number;
/**
* Tool time accumulated from closed tool batches during the current attempt. A batch spans the first dispatched call through the completion that drains the last outstanding one, so tools running concurrently within a turn are counted once.
*/
'live_tool_ms'?: number;
'tool_batch'?: StageToolBatchProjection | null;
'usage': BilledTokenCounts;
'model'?: BillingModelRef | null;
'todos'?: TodoListProjection | null;
@ -118,6 +130,10 @@ export interface StageProjection {
'mcp_servers'?: Array<McpServerProjection>;
'context_window'?: StageContextWindowProjection | null;
'inference'?: StageInferenceProjection | null;
/**
* Start of an external ACP agent process, if one is running. ACP agents do not expose Fabro\'s internal LLM brackets, so the process lifetime supplies their live inference estimate.
*/
'acp_started_at'?: string | null;
/**
* Whether the agent is executing normally or waiting for steering after an interrupt.
*/

View file

@ -15,12 +15,12 @@
/**
* Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`.
* Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. For a terminal stage these come from the worker\'s own stopwatch and are authoritative. For a stage still in flight they are a live estimate reconstructed from the event log, and `active_time_ms` is clamped to `wall_time_ms`. The estimate is replaced by the authoritative breakdown when the stage reaches a terminal event.
*/
export interface StageTiming {
'wall_time_ms': number;
'inference_time_ms'?: number;
'tool_time_ms'?: number;
'inference_time_ms': number;
'tool_time_ms': number;
/**
* Equals `inference_time_ms + tool_time_ms`.
*/

View file

@ -0,0 +1,33 @@
/* tslint:disable */
/* eslint-disable */
/**
* Fabro Run API
* HTTP API for managing Fabro workflow run executions.
*
* The version of the OpenAPI document: 0.1.0
*
*
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
* https://openapi-generator.tech
* Do not edit the class manually.
*/
/**
* One open tool batch: tool calls dispatched together that have not all reported completion. `open_call_ids` is a set rather than a count so a duplicated completion in a replayed log cannot drain the batch early.
*/
export interface StageToolBatchProjection {
/**
* Root agent session that dispatched the batch. Transitions are gated on it so delayed events from a replaced session cannot mutate the current batch.
*/
'session_id': string;
/**
* When the batch opened — the first dispatched call observed while no other calls were outstanding.
*/
'started_at': string;
/**
* Calls dispatched but not yet completed, by tool call id.
*/
'open_call_ids': Array<string>;
}

10
test/edge_only_node.fabro Normal file
View file

@ -0,0 +1,10 @@
digraph EdgeOnlyNode {
graph [goal="Reference a node that was never declared"]
/* `misspelled_node` is only ever named by an edge, never declared. */
start [shape=Mdiamond, label="Start"]
exit [shape=Msquare, label="Exit"]
start -> misspelled_node
misspelled_node -> exit
}

10
test/server-model.fabro Normal file
View file

@ -0,0 +1,10 @@
digraph ServerModel {
graph [goal="Use a server-owned model"]
start [shape=Mdiamond, label="Start"]
exit [shape=Msquare, label="Exit"]
work [label="Work", prompt="Do work", model="private-model", provider="server-only"]
start -> work -> exit
}