mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-10 03:30:59 +00:00
Merge remote-tracking branch 'origin/main' into codex/pr669-merge-main-3d8cf48
# Conflicts: # apps/fabro-web/app/routes/run-detail/model.ts
This commit is contained in:
commit
96604429af
88 changed files with 3803 additions and 1046 deletions
|
|
@ -11,10 +11,6 @@ leak-timeout = "500ms"
|
|||
filter = "package(fabro-server)"
|
||||
slow-timeout = { period = "5s", terminate-after = 4 }
|
||||
|
||||
[[profile.default.overrides]]
|
||||
filter = "package(fabro-server) & test(all_spec_routes_are_routable)"
|
||||
slow-timeout = { period = "15s", terminate-after = 4 }
|
||||
|
||||
[[profile.default.overrides]]
|
||||
filter = "package(fabro-workflow)"
|
||||
slow-timeout = { period = "2s", terminate-after = 3 }
|
||||
|
|
|
|||
|
|
@ -115,8 +115,10 @@ export function RunTableRow({
|
|||
</td>
|
||||
)}
|
||||
{show("size") && (
|
||||
<td className="whitespace-nowrap px-3 py-2.5 text-center">
|
||||
{run.size != null && <SizeChip size={run.size} />}
|
||||
<td className="relative z-10 px-3 py-2.5 text-center whitespace-nowrap">
|
||||
{run.size != null && (
|
||||
<SizeChip size={run.size} totalUsdMicros={run.totalUsdMicros} />
|
||||
)}
|
||||
</td>
|
||||
)}
|
||||
{show("changes") && (
|
||||
|
|
|
|||
58
apps/fabro-web/app/components/size-chip.test.tsx
Normal file
58
apps/fabro-web/app/components/size-chip.test.tsx
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
||||
import TestRenderer, { act } from "react-test-renderer";
|
||||
|
||||
import { setupReactTestEnv } from "../lib/test-utils";
|
||||
import { SizeChip } from "./size-chip";
|
||||
import { Tooltip } from "./ui";
|
||||
|
||||
let teardownReactTestEnv: (() => void) | undefined;
|
||||
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
|
||||
|
||||
function render(element: React.ReactElement): TestRenderer.ReactTestRenderer {
|
||||
let renderer: TestRenderer.ReactTestRenderer | undefined;
|
||||
act(() => {
|
||||
renderer = TestRenderer.create(element);
|
||||
});
|
||||
mountedRenderers.push(renderer!);
|
||||
return renderer!;
|
||||
}
|
||||
|
||||
function tooltipLabel(element: React.ReactElement): string {
|
||||
return render(element).root.findByType(Tooltip).props.label as string;
|
||||
}
|
||||
|
||||
describe("SizeChip", () => {
|
||||
beforeEach(() => {
|
||||
teardownReactTestEnv = setupReactTestEnv();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
act(() => {
|
||||
for (const renderer of mountedRenderers.splice(0)) {
|
||||
renderer.unmount();
|
||||
}
|
||||
});
|
||||
teardownReactTestEnv?.();
|
||||
teardownReactTestEnv = undefined;
|
||||
});
|
||||
|
||||
test("renders the size letter", () => {
|
||||
expect(JSON.stringify(render(<SizeChip size="M" />).toJSON())).toContain("M");
|
||||
});
|
||||
|
||||
test("appends the cost to the tooltip", () => {
|
||||
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={12_340_000} />))
|
||||
.toBe("Size M · $12.34");
|
||||
});
|
||||
|
||||
test("omits the cost when the run has no billing yet", () => {
|
||||
expect(tooltipLabel(<SizeChip size="M" />)).toBe("Size M");
|
||||
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={null} />)).toBe("Size M");
|
||||
});
|
||||
|
||||
test("calls out the tiers that warrant attention", () => {
|
||||
expect(tooltipLabel(<SizeChip size="L" totalUsdMicros={150_000_000} />))
|
||||
.toBe("Size L (risky) · $150.00");
|
||||
expect(tooltipLabel(<SizeChip size="XL" />)).toBe("Size XL (unhealthy)");
|
||||
});
|
||||
});
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
import { memo } from "react";
|
||||
import type { RunSize } from "@qltysh/fabro-api-client";
|
||||
|
||||
import { formatUsdMicros } from "../lib/format";
|
||||
|
|
@ -11,7 +12,7 @@ const SIZE_TONE: Record<RunSize, { className: string; note: string | null }> = {
|
|||
XL: { className: "bg-coral/15 text-coral", note: "unhealthy" },
|
||||
};
|
||||
|
||||
export function SizeChip({
|
||||
export const SizeChip = memo(function SizeChip({
|
||||
size,
|
||||
totalUsdMicros,
|
||||
}: {
|
||||
|
|
@ -19,10 +20,10 @@ export function SizeChip({
|
|||
totalUsdMicros?: number | null;
|
||||
}) {
|
||||
const tone = SIZE_TONE[size];
|
||||
const billed = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)} billed` : "";
|
||||
const amount = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)}` : "";
|
||||
const tooltip = tone.note != null
|
||||
? `Size ${size} (${tone.note})${billed}`
|
||||
: `Size ${size}${billed}`;
|
||||
? `Size ${size} (${tone.note})${amount}`
|
||||
: `Size ${size}${amount}`;
|
||||
return (
|
||||
<Tooltip label={tooltip}>
|
||||
<span className={`rounded px-1.5 py-0.5 font-mono text-xs font-bold tabular-nums ${tone.className}`}>
|
||||
|
|
@ -30,4 +31,4 @@ export function SizeChip({
|
|||
</span>
|
||||
</Tooltip>
|
||||
);
|
||||
}
|
||||
});
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ import { StagePopover } from "./stage-popover";
|
|||
import { deriveStageSummary } from "./stage-popover-summary";
|
||||
import type { Stage } from "../lib/stage-sidebar";
|
||||
import { generatedAxios } from "../lib/api-client";
|
||||
import { makeBilledTokenCounts } from "../lib/test-fixtures";
|
||||
|
||||
function makeEvent(overrides: Partial<EventEnvelope>): EventEnvelope {
|
||||
return {
|
||||
|
|
@ -34,6 +35,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
|
|||
duration: "1m 30s",
|
||||
startedAt: "2026-05-24T11:58:30Z",
|
||||
providerUsed: { mode: "policy", model: "claude-opus-4-7", reasoning_effort: "high" },
|
||||
billing: makeBilledTokenCounts(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import type { EventEnvelope } from "@qltysh/fabro-api-client";
|
|||
import TestRenderer, { act } from "react-test-renderer";
|
||||
|
||||
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
|
||||
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
|
||||
import type { Stage } from "../stage-sidebar";
|
||||
import { FanInResults } from "./fan-in-results";
|
||||
|
||||
|
|
@ -22,6 +23,7 @@ const fanInStage: Stage = {
|
|||
visit: 1,
|
||||
startedAt: "2026-04-09T12:00:00Z",
|
||||
providerUsed: null,
|
||||
billing: makeBilledTokenCounts(),
|
||||
};
|
||||
|
||||
function event(seq: number, partial: Partial<EventEnvelope>): EventEnvelope {
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import TestRenderer, { act } from "react-test-renderer";
|
|||
import { MemoryRouter } from "react-router";
|
||||
|
||||
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
|
||||
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
|
||||
import type { Stage } from "../stage-sidebar";
|
||||
import { ParallelChildren } from "./parallel-children";
|
||||
|
||||
|
|
@ -23,6 +24,7 @@ const parallelStage: Stage = {
|
|||
visit: 1,
|
||||
startedAt: "2026-04-09T12:00:00Z",
|
||||
providerUsed: null,
|
||||
billing: makeBilledTokenCounts(),
|
||||
};
|
||||
|
||||
function event(partial: Partial<EventEnvelope>): EventEnvelope {
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import { describe, expect, test } from "bun:test";
|
|||
import TestRenderer, { act } from "react-test-renderer";
|
||||
import { MemoryRouter } from "react-router";
|
||||
|
||||
import { makeBilledTokenCounts } from "../lib/test-fixtures";
|
||||
import { StageSidebar, type Stage } from "./stage-sidebar";
|
||||
|
||||
function makeStage(overrides: Partial<Stage> = {}): Stage {
|
||||
|
|
@ -17,6 +18,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
|
|||
duration: "--",
|
||||
startedAt: null,
|
||||
providerUsed: null,
|
||||
billing: makeBilledTokenCounts(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -103,6 +103,17 @@ describe("mapRunListItem", () => {
|
|||
|
||||
expect(mapRunListItem(summary).title).toBe("Untitled run");
|
||||
});
|
||||
|
||||
test("carries the billed total so the size chip can show it on hover", () => {
|
||||
expect(mapRunListItem(makeRun()).totalUsdMicros).toBe(500000);
|
||||
});
|
||||
|
||||
test("leaves the billed total undefined for runs without terminal billing", () => {
|
||||
expect(mapRunListItem(makeRun({ billing: null })).totalUsdMicros).toBeUndefined();
|
||||
expect(
|
||||
mapRunListItem(makeRun({ billing: { total_usd_micros: null } })).totalUsdMicros,
|
||||
).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("mapRunToRunItem", () => {
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ export interface RunItem {
|
|||
createdBy: Principal;
|
||||
lastEventAt?: string;
|
||||
size?: RunSize;
|
||||
totalUsdMicros?: number;
|
||||
}
|
||||
|
||||
export const columnStatuses = [
|
||||
|
|
@ -119,6 +120,7 @@ export function mapRunListItem(item: Run): RunItem {
|
|||
additions: item.diff?.additions,
|
||||
deletions: item.diff?.deletions,
|
||||
size: item.size,
|
||||
totalUsdMicros: item.billing?.total_usd_micros ?? undefined,
|
||||
};
|
||||
}
|
||||
|
||||
|
|
|
|||
28
apps/fabro-web/app/lib/billing.ts
Normal file
28
apps/fabro-web/app/lib/billing.ts
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
import type { BilledTokenCounts } from "@qltysh/fabro-api-client";
|
||||
|
||||
export interface BillingTokenBucket {
|
||||
label: string;
|
||||
value: number;
|
||||
}
|
||||
|
||||
export function billableOutputTokens(billing: BilledTokenCounts): number {
|
||||
return billing.output_tokens + billing.reasoning_tokens;
|
||||
}
|
||||
|
||||
/** The disjoint token buckets shown in every billing breakdown. */
|
||||
export function billingTokenBuckets(billing: BilledTokenCounts): BillingTokenBucket[] {
|
||||
return [
|
||||
{ label: "Cache read", value: billing.cache_read_tokens },
|
||||
{ label: "Cache creation", value: billing.cache_write_tokens },
|
||||
{ label: "Uncached", value: billing.input_tokens },
|
||||
{ label: "Output", value: billableOutputTokens(billing) },
|
||||
];
|
||||
}
|
||||
|
||||
export function hasBillingUsage(billing: BilledTokenCounts): boolean {
|
||||
return (
|
||||
billing.total_tokens !== 0 ||
|
||||
(billing.total_usd_micros ?? 0) !== 0 ||
|
||||
billingTokenBuckets(billing).some((bucket) => bucket.value !== 0)
|
||||
);
|
||||
}
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
import { useCallback } from "react";
|
||||
import useSWR, { type SWRConfiguration } from "swr";
|
||||
import type {
|
||||
ApiQuestion,
|
||||
|
|
@ -73,6 +74,7 @@ import {
|
|||
type RunFileSelection,
|
||||
type RunGraphDirection,
|
||||
} from "./query-keys";
|
||||
import { isTerminalRunStatus } from "./run-actions";
|
||||
|
||||
const immutableOptions: SWRConfiguration = {
|
||||
revalidateIfStale: false,
|
||||
|
|
@ -179,10 +181,21 @@ export function useRunsPage(opts: RunsPageOptions = {}, enabled = true) {
|
|||
);
|
||||
}
|
||||
|
||||
export function useRun(id: string | undefined) {
|
||||
export function useRun(id: string | undefined, refreshInterval?: number) {
|
||||
const pollingInterval = useCallback(
|
||||
(run: Run | null | undefined) =>
|
||||
refreshInterval &&
|
||||
run?.timestamps.started_at &&
|
||||
!isTerminalRunStatus(run.lifecycle.status.kind)
|
||||
? refreshInterval
|
||||
: 0,
|
||||
[refreshInterval],
|
||||
);
|
||||
|
||||
return useSWR<Run | null>(
|
||||
id ? queryKeys.runs.detail(id) : null,
|
||||
() => apiNullableData(() => runsApi.retrieveRun(id!)),
|
||||
refreshInterval ? { refreshInterval: pollingInterval } : undefined,
|
||||
);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,6 @@
|
|||
import { describe, expect, test } from "bun:test";
|
||||
|
||||
import { queryKeys } from "./query-keys";
|
||||
import { queryKeysForRunEvent } from "./run-events";
|
||||
|
||||
describe("queryKeys", () => {
|
||||
test("uses semantic tuples as stable SWR keys and keeps SSE URLs explicit", () => {
|
||||
|
|
@ -61,52 +60,4 @@ describe("queryKeys", () => {
|
|||
expect(queryKeys.runs.attachUrl("run 1")).toBe("/api/v1/runs/run%201/attach");
|
||||
});
|
||||
|
||||
test("event-mapped keys match query hook resources", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "checkpoint.completed")).toEqual(
|
||||
[
|
||||
...queryKeys.runs.filesAllScopes("run-1"),
|
||||
queryKeys.runs.commits("run-1"),
|
||||
],
|
||||
);
|
||||
expect(queryKeysForRunEvent("run-1", "stage.completed", "stage-1")).toEqual([
|
||||
queryKeys.runs.stages("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.events("run-1", 1000),
|
||||
queryKeys.runs.graph("run-1", "LR"),
|
||||
queryKeys.runs.graph("run-1", "TB"),
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "stage-1"),
|
||||
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
|
||||
]);
|
||||
expect(queryKeysForRunEvent("run-1", "run.title.updated")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
]);
|
||||
});
|
||||
|
||||
test("agent activity events invalidate per-stage resources", () => {
|
||||
for (const event of [
|
||||
"stage.prompt",
|
||||
"agent.tool.started",
|
||||
"agent.tool.completed",
|
||||
"command.started",
|
||||
"command.completed",
|
||||
]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "stage-1")).toEqual([
|
||||
queryKeys.runs.stageEvents("run-1", "stage-1"),
|
||||
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
|
||||
]);
|
||||
}
|
||||
expect(queryKeysForRunEvent("run-1", "agent.message", "stage-1")).toEqual([
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "stage-1"),
|
||||
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
|
||||
]);
|
||||
});
|
||||
|
||||
test("agent message without a node_id still invalidates projected state", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "agent.message")).toEqual([
|
||||
queryKeys.runs.state("run-1"),
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -111,10 +111,7 @@ export async function deleteRuns(
|
|||
request?: Request,
|
||||
): Promise<BatchDeleteRunsResponse> {
|
||||
try {
|
||||
// See `batchRunLifecycleAction` for the `as unknown as` rationale:
|
||||
// openapi-generator types `uniqueItems` arrays as `Set<T>` while the wire
|
||||
// contract is a JSON array.
|
||||
const body = { run_ids: runIds, force } as unknown as BatchDeleteRunsRequest;
|
||||
const body: BatchDeleteRunsRequest = { run_ids: runIds, force };
|
||||
return await apiData(() => runsApi.batchDeleteRuns(body, requestSignalOptions(request)));
|
||||
} catch (error) {
|
||||
throw lifecycleActionErrorFromError(error);
|
||||
|
|
@ -284,10 +281,7 @@ async function batchRunLifecycleAction(
|
|||
request?: Request,
|
||||
): Promise<BatchRunLifecycleResponse> {
|
||||
try {
|
||||
// openapi-generator's TypeScript client represents `uniqueItems` arrays as
|
||||
// Set<T>, but the HTTP wire contract is still a JSON array. Keep an array
|
||||
// here so Axios serializes the request body correctly.
|
||||
const body = { run_ids: runIds } as unknown as BatchRunLifecycleRequest;
|
||||
const body: BatchRunLifecycleRequest = { run_ids: runIds };
|
||||
switch (action) {
|
||||
case "archive":
|
||||
return await apiData(() => runsApi.batchArchiveRuns(body, requestSignalOptions(request)));
|
||||
|
|
|
|||
|
|
@ -85,6 +85,8 @@ describe("queryKeysForRunEvent", () => {
|
|||
|
||||
test("interrupt settlement invalidates projected control state and stage activity", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "agent.round.interrupted", "nap@1")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.events("run-1", 1000),
|
||||
queryKeys.runs.stageEvents("run-1", "nap@1"),
|
||||
|
|
@ -129,10 +131,16 @@ describe("queryKeysForRunEvent", () => {
|
|||
test("every inference projection transition invalidates live run state", () => {
|
||||
for (const event of [
|
||||
"agent.llm.started",
|
||||
"agent.llm.first_output",
|
||||
"agent.llm.retry",
|
||||
"agent.error",
|
||||
]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
]);
|
||||
}
|
||||
for (const event of ["agent.llm.first_output", "agent.llm.retry"]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
|
|
@ -141,15 +149,47 @@ describe("queryKeysForRunEvent", () => {
|
|||
expect(
|
||||
queryKeysForRunEvent("run-1", "agent.message", "code@1"),
|
||||
).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
queryKeys.runs.stageContextWindow("run-1", "code@1"),
|
||||
]);
|
||||
expect(queryKeysForRunEvent("run-1", "agent.session.ended")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
]);
|
||||
});
|
||||
|
||||
test("ACP timing events invalidate live summaries and stage events", () => {
|
||||
for (const event of [
|
||||
"agent.acp.started",
|
||||
"agent.acp.completed",
|
||||
"agent.acp.cancelled",
|
||||
"agent.acp.timed_out",
|
||||
]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
]);
|
||||
}
|
||||
});
|
||||
|
||||
test("tool timing events invalidate live summaries and stage resources", () => {
|
||||
for (const event of ["agent.tool.started", "agent.tool.completed"]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.billing("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
queryKeys.runs.stageContextWindow("run-1", "code@1"),
|
||||
]);
|
||||
}
|
||||
});
|
||||
|
||||
test("watchdog timeout refreshes the stage events for that stage", () => {
|
||||
expect(
|
||||
queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"),
|
||||
|
|
|
|||
|
|
@ -118,6 +118,22 @@ const INFERENCE_EVENTS = new Set([
|
|||
"agent.error",
|
||||
"agent.session.ended",
|
||||
]);
|
||||
const INFERENCE_TIMING_EVENTS = new Set([
|
||||
"agent.llm.started",
|
||||
"agent.message",
|
||||
"agent.error",
|
||||
"agent.session.ended",
|
||||
]);
|
||||
const TOOL_TIMING_EVENTS = new Set([
|
||||
"agent.tool.started",
|
||||
"agent.tool.completed",
|
||||
]);
|
||||
const ACP_TIMING_EVENTS = new Set([
|
||||
"agent.acp.started",
|
||||
"agent.acp.completed",
|
||||
"agent.acp.cancelled",
|
||||
"agent.acp.timed_out",
|
||||
]);
|
||||
// Todo / task mutation events refresh `getRunState` consumers (so per-stage
|
||||
// todo projections update live) and the run events list.
|
||||
const TODO_EVENTS = new Set([
|
||||
|
|
@ -126,6 +142,14 @@ const TODO_EVENTS = new Set([
|
|||
"todo.deleted",
|
||||
]);
|
||||
|
||||
function liveTimingKeys(runId: string): Key[] {
|
||||
return [
|
||||
queryKeys.runs.detail(runId),
|
||||
queryKeys.runs.state(runId),
|
||||
queryKeys.runs.billing(runId),
|
||||
];
|
||||
}
|
||||
|
||||
export function queryKeysForRunEvent(
|
||||
runId: string,
|
||||
event: string,
|
||||
|
|
@ -184,6 +208,12 @@ export function queryKeysForRunEvent(
|
|||
if (AGENT_CONTROL_STATE_EVENTS.has(event)) {
|
||||
keys.unshift(queryKeys.runs.state(runId));
|
||||
}
|
||||
if (event === "agent.round.interrupted") {
|
||||
keys.unshift(
|
||||
queryKeys.runs.detail(runId),
|
||||
queryKeys.runs.billing(runId),
|
||||
);
|
||||
}
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
|
||||
|
|
@ -192,7 +222,9 @@ export function queryKeysForRunEvent(
|
|||
}
|
||||
|
||||
if (INFERENCE_EVENTS.has(event)) {
|
||||
const keys: Key[] = [queryKeys.runs.state(runId)];
|
||||
const keys = INFERENCE_TIMING_EVENTS.has(event)
|
||||
? liveTimingKeys(runId)
|
||||
: [queryKeys.runs.state(runId)];
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
if (event === "agent.message") {
|
||||
|
|
@ -202,6 +234,23 @@ export function queryKeysForRunEvent(
|
|||
return keys;
|
||||
}
|
||||
|
||||
if (TOOL_TIMING_EVENTS.has(event)) {
|
||||
const keys = liveTimingKeys(runId);
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
if (ACP_TIMING_EVENTS.has(event)) {
|
||||
const keys = liveTimingKeys(runId);
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
if (event === "watchdog.timeout") {
|
||||
return stageId ? [queryKeys.runs.stageEvents(runId, stageId)] : [];
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import type { PaginatedRunStageList, StageHandler, StageState } from "@qltysh/fa
|
|||
|
||||
import type { Stage } from "../components/stage-sidebar";
|
||||
import { aggregateGraphNodeStatus, formatStageLabel, mapRunStagesToSidebarStages } from "./stage-sidebar";
|
||||
import { makeBilledTokenCounts } from "./test-fixtures";
|
||||
|
||||
function makeStage(nodeId: string, visit: number, status: StageState): Stage {
|
||||
return {
|
||||
|
|
@ -17,6 +18,7 @@ function makeStage(nodeId: string, visit: number, status: StageState): Stage {
|
|||
duration: "--",
|
||||
startedAt: null,
|
||||
providerUsed: null,
|
||||
billing: makeBilledTokenCounts(),
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -38,6 +40,15 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
model: "gpt-5.5",
|
||||
reasoning_effort: "high",
|
||||
},
|
||||
billing: makeBilledTokenCounts({
|
||||
input_tokens: 28_640,
|
||||
output_tokens: 7_550,
|
||||
total_tokens: 43_690,
|
||||
reasoning_tokens: 1_200,
|
||||
cache_read_tokens: 4_800,
|
||||
cache_write_tokens: 1_500,
|
||||
total_usd_micros: 720_000,
|
||||
}),
|
||||
},
|
||||
{
|
||||
id: "apply-changes@2",
|
||||
|
|
@ -46,6 +57,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "running",
|
||||
node_id: "apply",
|
||||
visit: 2,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
@ -64,6 +76,10 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
model: "gpt-5.5",
|
||||
reasoning_effort: "high",
|
||||
});
|
||||
// Each visit keeps its own tokens and cost, so the stage popover never
|
||||
// shows a sibling visit's usage.
|
||||
expect(result[0].billing.total_usd_micros).toBe(720_000);
|
||||
expect(result[1].billing.total_usd_micros).toBeUndefined();
|
||||
expect(formatStageLabel(result[0])).toBe("Apply Changes");
|
||||
|
||||
expect(result[1].id).toBe("apply-changes@2");
|
||||
|
|
@ -83,6 +99,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "succeeded",
|
||||
node_id: "start",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
{
|
||||
id: "verify@1",
|
||||
|
|
@ -91,6 +108,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "succeeded",
|
||||
node_id: "verify",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
{
|
||||
id: "exit@1",
|
||||
|
|
@ -99,6 +117,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "succeeded",
|
||||
node_id: "exit",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
@ -118,6 +137,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "running",
|
||||
node_id: "verify",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
@ -137,6 +157,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
node_id: "work",
|
||||
visit: 1,
|
||||
graph_visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
{
|
||||
id: "work@2",
|
||||
|
|
@ -147,6 +168,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
visit: 2,
|
||||
graph_visit: 1,
|
||||
resumed_from_stage_id: "work@1",
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
@ -172,6 +194,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "succeeded",
|
||||
node_id: "verify",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
@ -192,6 +215,7 @@ describe("mapRunStagesToSidebarStages", () => {
|
|||
status: "pending",
|
||||
node_id: "approval",
|
||||
visit: 1,
|
||||
billing: makeBilledTokenCounts(),
|
||||
},
|
||||
],
|
||||
meta: { has_more: false },
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import { StageState } from "@qltysh/fabro-api-client";
|
||||
import type {
|
||||
BilledTokenCounts,
|
||||
PaginatedRunStageList,
|
||||
StageHandler,
|
||||
StageModelUsage,
|
||||
|
|
@ -27,6 +28,11 @@ export interface Stage {
|
|||
resumedFromStageId: string | null;
|
||||
startedAt: string | null;
|
||||
providerUsed: StageModelUsage | null;
|
||||
/**
|
||||
* Tokens and cost for this visit alone, priced the same way the Billing tab
|
||||
* prices its per-node rows. All-zero counts mean the stage called no model.
|
||||
*/
|
||||
billing: BilledTokenCounts;
|
||||
}
|
||||
|
||||
export const ACTIVE_STAGE_STATES: ReadonlySet<StageState> = new Set([
|
||||
|
|
@ -102,6 +108,7 @@ export function mapRunStagesToSidebarStages(
|
|||
: "--",
|
||||
startedAt: stage.started_at ?? null,
|
||||
providerUsed: stage.provider_used ?? null,
|
||||
billing: stage.billing,
|
||||
});
|
||||
}
|
||||
return stages;
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import type { Principal } from "@qltysh/fabro-api-client";
|
||||
import type { BilledTokenCounts, Principal } from "@qltysh/fabro-api-client";
|
||||
|
||||
export const TEST_PRINCIPAL: Principal = {
|
||||
kind: "user",
|
||||
|
|
@ -6,3 +6,17 @@ export const TEST_PRINCIPAL: Principal = {
|
|||
login: "test",
|
||||
auth_method: "dev_token",
|
||||
};
|
||||
|
||||
export function makeBilledTokenCounts(
|
||||
overrides: Partial<BilledTokenCounts> = {},
|
||||
): BilledTokenCounts {
|
||||
return {
|
||||
cache_read_tokens: 0,
|
||||
cache_write_tokens: 0,
|
||||
input_tokens: 0,
|
||||
output_tokens: 0,
|
||||
reasoning_tokens: 0,
|
||||
total_tokens: 0,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,14 +1,18 @@
|
|||
import { useMemo } from "react";
|
||||
import { useParams } from "react-router";
|
||||
import { ArrowDownTrayIcon, PaperClipIcon } from "@heroicons/react/24/outline";
|
||||
import { Disclosure, DisclosureButton, DisclosurePanel } from "@headlessui/react";
|
||||
import { ArrowDownTrayIcon, ChevronRightIcon, PaperClipIcon } from "@heroicons/react/24/outline";
|
||||
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
|
||||
|
||||
import { EmptyState, ErrorState, LoadingState } from "../components/state";
|
||||
import { StageSidebar } from "../components/stage-sidebar";
|
||||
import { stageArtifactDownloadUrl } from "../lib/api-client";
|
||||
import { formatBytes } from "../lib/format";
|
||||
import { plural } from "../lib/plural";
|
||||
import { useRunArtifacts, useRunStages } from "../lib/queries";
|
||||
import { formatStageLabel, mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
|
||||
import { mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
|
||||
import type { ArtifactFile, ArtifactVersion } from "./run-artifacts/group";
|
||||
import { groupArtifactsByFile } from "./run-artifacts/group";
|
||||
|
||||
export const handle = { wide: true };
|
||||
|
||||
|
|
@ -25,7 +29,12 @@ export default function RunArtifacts() {
|
|||
<div className="flex gap-6">
|
||||
<StageSidebar stages={stages} runId={id!} activeLink="artifacts" />
|
||||
<div className="min-w-0 flex-1">
|
||||
<RunArtifactsBody runId={id!} artifactsQuery={artifactsQuery} stages={stages} />
|
||||
<RunArtifactsBody
|
||||
runId={id!}
|
||||
artifactsQuery={artifactsQuery}
|
||||
stagesQuery={stagesQuery}
|
||||
stages={stages}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
|
@ -34,86 +43,40 @@ export default function RunArtifacts() {
|
|||
function RunArtifactsBody({
|
||||
runId,
|
||||
artifactsQuery,
|
||||
stagesQuery,
|
||||
stages,
|
||||
}: {
|
||||
runId: string;
|
||||
artifactsQuery: ReturnType<typeof useRunArtifacts>;
|
||||
stagesQuery: ReturnType<typeof useRunStages>;
|
||||
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
|
||||
}) {
|
||||
if (artifactsQuery.error) {
|
||||
const error = artifactsQuery.error ?? stagesQuery.error;
|
||||
if (error) {
|
||||
return (
|
||||
<ErrorState
|
||||
title="Couldn't load artifacts"
|
||||
description={errorMessage(artifactsQuery.error)}
|
||||
onRetry={() => void artifactsQuery.mutate()}
|
||||
description={errorMessage(error)}
|
||||
onRetry={() => {
|
||||
if (artifactsQuery.error) void artifactsQuery.mutate();
|
||||
if (stagesQuery.error) void stagesQuery.mutate();
|
||||
}}
|
||||
/>
|
||||
);
|
||||
}
|
||||
if (artifactsQuery.data === undefined) {
|
||||
if (artifactsQuery.data === undefined || stagesQuery.data === undefined) {
|
||||
return <LoadingState label="Loading artifacts…" />;
|
||||
}
|
||||
const entries = artifactsQuery.data?.data ?? [];
|
||||
if (entries.length === 0) {
|
||||
return (
|
||||
<EmptyState
|
||||
icon={PaperClipIcon}
|
||||
title="No artifacts captured"
|
||||
description="No stage in this run produced any artifacts."
|
||||
/>
|
||||
);
|
||||
}
|
||||
return <ArtifactList runId={runId} entries={entries} stages={stages} />;
|
||||
return (
|
||||
<ArtifactFiles
|
||||
runId={runId}
|
||||
entries={artifactsQuery.data?.data ?? []}
|
||||
stages={stages}
|
||||
/>
|
||||
);
|
||||
}
|
||||
|
||||
interface StageGroup {
|
||||
key: string;
|
||||
stageId: string;
|
||||
retry: number;
|
||||
label: string;
|
||||
entries: RunArtifactEntry[];
|
||||
totalBytes: number;
|
||||
}
|
||||
|
||||
function groupArtifacts(
|
||||
entries: readonly RunArtifactEntry[],
|
||||
stages: ReturnType<typeof mapRunStagesToSidebarStages>,
|
||||
): StageGroup[] {
|
||||
const stageLabels = new Map<string, string>();
|
||||
for (const stage of stages) {
|
||||
stageLabels.set(stage.id, formatStageLabel(stage));
|
||||
}
|
||||
|
||||
const groups = new Map<string, StageGroup>();
|
||||
for (const entry of entries) {
|
||||
const key = `${entry.stage_id}#${entry.retry}`;
|
||||
const existing = groups.get(key);
|
||||
if (existing) {
|
||||
existing.entries.push(entry);
|
||||
existing.totalBytes += entry.size;
|
||||
} else {
|
||||
groups.set(key, {
|
||||
key,
|
||||
stageId: entry.stage_id,
|
||||
retry: entry.retry,
|
||||
label: stageLabels.get(entry.stage_id) ?? entry.node_slug,
|
||||
entries: [entry],
|
||||
totalBytes: entry.size,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
for (const group of groups.values()) {
|
||||
group.entries.sort((a, b) => a.relative_path.localeCompare(b.relative_path));
|
||||
}
|
||||
const sortedGroups = Array.from(groups.values());
|
||||
sortedGroups.sort((a, b) => {
|
||||
const labelCmp = a.label.localeCompare(b.label);
|
||||
return labelCmp !== 0 ? labelCmp : a.retry - b.retry;
|
||||
});
|
||||
return sortedGroups;
|
||||
}
|
||||
|
||||
function ArtifactList({
|
||||
function ArtifactFiles({
|
||||
runId,
|
||||
entries,
|
||||
stages,
|
||||
|
|
@ -122,97 +85,209 @@ function ArtifactList({
|
|||
entries: readonly RunArtifactEntry[];
|
||||
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
|
||||
}) {
|
||||
const groups = useMemo(() => groupArtifacts(entries, stages), [entries, stages]);
|
||||
const totalBytes = useMemo(
|
||||
() => entries.reduce((sum, entry) => sum + entry.size, 0),
|
||||
[entries],
|
||||
);
|
||||
const files = useMemo(() => groupArtifactsByFile(entries, stages), [entries, stages]);
|
||||
|
||||
if (files.length === 0) {
|
||||
return (
|
||||
<EmptyState
|
||||
icon={PaperClipIcon}
|
||||
title="No artifacts captured"
|
||||
description="No stage in this run produced any artifacts."
|
||||
/>
|
||||
);
|
||||
}
|
||||
return <ArtifactList runId={runId} files={files} />;
|
||||
}
|
||||
|
||||
function ArtifactList({ runId, files }: { runId: string; files: readonly ArtifactFile[] }) {
|
||||
const { captures, latestBytes, storedBytes } = useMemo(() => {
|
||||
let captures = 0;
|
||||
let latestBytes = 0;
|
||||
let storedBytes = 0;
|
||||
for (const file of files) {
|
||||
captures += file.versions.length;
|
||||
latestBytes += file.versions[0].size;
|
||||
for (const version of file.versions) storedBytes += version.size;
|
||||
}
|
||||
return { captures, latestBytes, storedBytes };
|
||||
}, [files]);
|
||||
|
||||
// Only mention versions once some file actually has more than one.
|
||||
const versioned = captures > files.length;
|
||||
|
||||
return (
|
||||
<div className="space-y-4">
|
||||
<div className="flex items-baseline justify-between">
|
||||
<div className="flex flex-wrap items-baseline justify-between gap-4">
|
||||
<h2 className="text-sm font-medium text-fg">
|
||||
{entries.length} {entries.length === 1 ? "artifact" : "artifacts"}
|
||||
{files.length} {plural(files.length, "file", "files")}
|
||||
{versioned && (
|
||||
<span className="font-normal text-fg-muted">
|
||||
{" "}
|
||||
· {captures} {plural(captures, "version", "versions")}
|
||||
</span>
|
||||
)}
|
||||
</h2>
|
||||
<span className="text-xs tabular-nums text-fg-muted">
|
||||
{formatBytes(totalBytes)} total
|
||||
<span className="text-xs text-fg-muted tabular-nums">
|
||||
{versioned
|
||||
? `${formatBytes(latestBytes)} latest · ${formatBytes(storedBytes)} stored`
|
||||
: `${formatBytes(latestBytes)} total`}
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{groups.map((group) => (
|
||||
<StageGroupCard key={group.key} runId={runId} group={group} />
|
||||
))}
|
||||
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
|
||||
{files.map((file) => (
|
||||
<ArtifactFileRow key={file.path} runId={runId} file={file} />
|
||||
))}
|
||||
</section>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function StageGroupCard({ runId, group }: { runId: string; group: StageGroup }) {
|
||||
function ArtifactFileRow({ runId, file }: { runId: string; file: ArtifactFile }) {
|
||||
const hasEarlier = file.versions.length > 1;
|
||||
const latest = file.versions[0];
|
||||
|
||||
return (
|
||||
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
|
||||
<header className="flex items-baseline justify-between border-b border-line px-4 py-2.5">
|
||||
<div className="flex items-baseline gap-2">
|
||||
<h3 className="text-sm font-medium text-fg">{group.label}</h3>
|
||||
{group.retry > 0 && (
|
||||
<span className="rounded bg-overlay px-1.5 py-0.5 text-[11px] font-medium text-fg-3">
|
||||
retry {group.retry}
|
||||
<Disclosure as="div" className="border-t border-line first:border-t-0">
|
||||
{({ open }) => (
|
||||
<>
|
||||
<div className="flex items-center gap-2 px-3 py-2.5 sm:gap-4 sm:px-4">
|
||||
{hasEarlier ? (
|
||||
<DisclosureButton className="group shrink-0 rounded-md p-1 text-fg-3 transition-colors hover:bg-overlay hover:text-fg-2 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500">
|
||||
<span className="sr-only">
|
||||
{open ? "Hide" : "Show"} earlier versions of {file.name}
|
||||
</span>
|
||||
<ChevronRightIcon
|
||||
className="size-3.5 transition-transform group-data-open:rotate-90"
|
||||
aria-hidden="true"
|
||||
/>
|
||||
</DisclosureButton>
|
||||
) : (
|
||||
<span className="size-5 shrink-0" aria-hidden="true" />
|
||||
)}
|
||||
|
||||
<span className="min-w-0 flex-1" title={file.path}>
|
||||
<span className="block truncate font-mono text-xs">
|
||||
<span className="text-fg-muted">{file.dir}</span>
|
||||
<span className="text-fg-2">{file.name}</span>
|
||||
</span>
|
||||
<span className="mt-0.5 block truncate text-[11px] text-fg-3 md:hidden">
|
||||
<VersionLabel version={latest} />
|
||||
</span>
|
||||
</span>
|
||||
|
||||
{hasEarlier && (
|
||||
<span className="hidden shrink-0 rounded-full bg-overlay-strong px-2 py-0.5 text-[11px] text-fg-3 lg:inline">
|
||||
{file.versions.length}{" "}
|
||||
{plural(file.versions.length, "version", "versions")}
|
||||
</span>
|
||||
)}
|
||||
|
||||
<span className="hidden max-w-48 shrink-0 truncate text-xs text-fg-3 md:inline">
|
||||
<VersionLabel version={latest} />
|
||||
</span>
|
||||
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
|
||||
{formatBytes(latest.size)}
|
||||
</span>
|
||||
<DownloadLink runId={runId} file={file} version={latest} />
|
||||
</div>
|
||||
|
||||
{hasEarlier && (
|
||||
<DisclosurePanel
|
||||
as="ul"
|
||||
className="border-t border-line bg-black/15 py-1"
|
||||
>
|
||||
<EarlierVersions runId={runId} file={file} />
|
||||
</DisclosurePanel>
|
||||
)}
|
||||
</div>
|
||||
<span className="text-xs tabular-nums text-fg-muted">
|
||||
{group.entries.length} {group.entries.length === 1 ? "file" : "files"}
|
||||
{" · "}
|
||||
{formatBytes(group.totalBytes)}
|
||||
</span>
|
||||
</header>
|
||||
<ul className="divide-y divide-line">
|
||||
{group.entries.map((entry) => (
|
||||
<ArtifactRow
|
||||
key={`${group.key}#${entry.relative_path}`}
|
||||
runId={runId}
|
||||
entry={entry}
|
||||
/>
|
||||
))}
|
||||
</ul>
|
||||
</section>
|
||||
</>
|
||||
)}
|
||||
</Disclosure>
|
||||
);
|
||||
}
|
||||
|
||||
function ArtifactRow({ runId, entry }: { runId: string; entry: RunArtifactEntry }) {
|
||||
function EarlierVersions({ runId, file }: { runId: string; file: ArtifactFile }) {
|
||||
return (
|
||||
<>
|
||||
{file.versions.map((version, index) =>
|
||||
index === 0 ? null : (
|
||||
<li
|
||||
key={`${version.stageId}#${version.retry}`}
|
||||
className="flex items-center gap-2 py-1.5 pr-3 pl-10 hover:bg-overlay sm:gap-4 sm:pr-4 sm:pl-14"
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate text-xs text-fg-3">
|
||||
<VersionLabel version={version} />
|
||||
</span>
|
||||
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
|
||||
{formatBytes(version.size)}
|
||||
</span>
|
||||
<SizeDelta delta={version.delta} />
|
||||
<DownloadLink runId={runId} file={file} version={version} />
|
||||
</li>
|
||||
),
|
||||
)}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
function VersionLabel({ version }: { version: ArtifactVersion }) {
|
||||
const attempt = attemptLabel(version);
|
||||
return (
|
||||
<>
|
||||
{version.stageLabel}
|
||||
{attempt && <span className="ml-2 text-fg-muted">{attempt}</span>}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
function attemptLabel(version: ArtifactVersion): string | null {
|
||||
return version.retry > 1 ? `attempt ${version.retry}` : null;
|
||||
}
|
||||
|
||||
function SizeDelta({ delta }: { delta: number | null }) {
|
||||
if (delta === null) {
|
||||
return <span className="shrink-0 text-[11px] text-fg-muted tabular-nums">first</span>;
|
||||
}
|
||||
const tone = delta < 0 ? "text-amber" : "text-mint";
|
||||
const sign = delta < 0 ? "−" : "+";
|
||||
return (
|
||||
<span className={`shrink-0 text-[11px] ${tone} tabular-nums`}>
|
||||
{sign}
|
||||
{formatBytes(Math.abs(delta))}
|
||||
</span>
|
||||
);
|
||||
}
|
||||
|
||||
function DownloadLink({
|
||||
runId,
|
||||
file,
|
||||
version,
|
||||
}: {
|
||||
runId: string;
|
||||
file: ArtifactFile;
|
||||
version: ArtifactVersion;
|
||||
}) {
|
||||
const href = stageArtifactDownloadUrl(
|
||||
runId,
|
||||
entry.stage_id,
|
||||
entry.relative_path,
|
||||
entry.retry,
|
||||
version.stageId,
|
||||
file.path,
|
||||
version.retry,
|
||||
);
|
||||
|
||||
const attempt = attemptLabel(version);
|
||||
const source = attempt ? `${version.stageLabel}, ${attempt}` : version.stageLabel;
|
||||
return (
|
||||
<li className="flex items-center gap-4 px-4 py-2">
|
||||
<span
|
||||
className="flex-1 truncate font-mono text-xs text-fg-2"
|
||||
title={entry.relative_path}
|
||||
>
|
||||
{entry.relative_path}
|
||||
</span>
|
||||
<span className="shrink-0 tabular-nums text-xs text-fg-muted">
|
||||
{formatBytes(entry.size)}
|
||||
</span>
|
||||
<a
|
||||
href={href}
|
||||
download={basename(entry.relative_path)}
|
||||
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
|
||||
>
|
||||
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
|
||||
Download
|
||||
</a>
|
||||
</li>
|
||||
<a
|
||||
href={href}
|
||||
download={file.name}
|
||||
aria-label={`Download ${file.name} from ${source}`}
|
||||
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
|
||||
>
|
||||
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
|
||||
<span className="hidden sm:inline">Download</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
|
||||
function basename(path: string): string {
|
||||
const idx = path.lastIndexOf("/");
|
||||
return idx >= 0 ? path.slice(idx + 1) : path;
|
||||
}
|
||||
|
||||
function errorMessage(error: unknown): string | undefined {
|
||||
return error instanceof Error ? error.message : undefined;
|
||||
}
|
||||
|
|
|
|||
192
apps/fabro-web/app/routes/run-artifacts/group.test.ts
Normal file
192
apps/fabro-web/app/routes/run-artifacts/group.test.ts
Normal file
|
|
@ -0,0 +1,192 @@
|
|||
import { describe, expect, test } from "bun:test";
|
||||
import { StageHandler, StageState } from "@qltysh/fabro-api-client";
|
||||
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
|
||||
|
||||
import type { Stage } from "../../lib/stage-sidebar";
|
||||
import { groupArtifactsByFile, splitArtifactPath } from "./group";
|
||||
|
||||
function stage(nodeId: string, startedAt: string | null, visit = 1): Stage {
|
||||
return {
|
||||
id: `${nodeId}@${visit}`,
|
||||
name: nodeId,
|
||||
handler: StageHandler.AGENT,
|
||||
nodeId,
|
||||
visit,
|
||||
graphVisit: null,
|
||||
resumedFromStageId: null,
|
||||
status: StageState.SUCCEEDED,
|
||||
duration: "1s",
|
||||
startedAt,
|
||||
providerUsed: null,
|
||||
};
|
||||
}
|
||||
|
||||
function artifact(
|
||||
nodeSlug: string,
|
||||
path: string,
|
||||
size: number,
|
||||
retry = 1,
|
||||
visit = 1,
|
||||
): RunArtifactEntry {
|
||||
return {
|
||||
stage_id: `${nodeSlug}@${visit}`,
|
||||
node_slug: nodeSlug,
|
||||
retry,
|
||||
relative_path: path,
|
||||
size,
|
||||
};
|
||||
}
|
||||
|
||||
/** Mirrors run 01KYJ8ZR0N: one report rewritten by four stages. */
|
||||
const REPORT = ".ai/reports/2026-07-27-wrk-002-instance-lifecycle.md";
|
||||
|
||||
const STAGES: Stage[] = [
|
||||
stage("start", "2026-07-27T17:12:08Z"),
|
||||
stage("plan", "2026-07-27T17:21:44Z"),
|
||||
stage("implement_plan", "2026-07-27T17:44:18Z"),
|
||||
stage("simplify", "2026-07-27T18:43:11Z"),
|
||||
stage("consolidate_reviews", "2026-07-27T19:44:26Z"),
|
||||
stage("fix_review_findings", "2026-07-27T19:49:22Z"),
|
||||
];
|
||||
|
||||
describe("splitArtifactPath", () => {
|
||||
test("splits a nested path into directory prefix and filename", () => {
|
||||
expect(splitArtifactPath(".ai/reports/run.md")).toEqual({
|
||||
dir: ".ai/reports/",
|
||||
name: "run.md",
|
||||
});
|
||||
});
|
||||
|
||||
test("leaves a root-level path without a directory", () => {
|
||||
expect(splitArtifactPath("README.md")).toEqual({ dir: "", name: "README.md" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("groupArtifactsByFile", () => {
|
||||
test("collapses repeated captures of one path into a single file", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("consolidate_reviews", REPORT, 14323),
|
||||
artifact("fix_review_findings", REPORT, 17483),
|
||||
artifact("implement_plan", REPORT, 8422),
|
||||
artifact("simplify", REPORT, 13162),
|
||||
],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
expect(files).toHaveLength(1);
|
||||
expect(files[0].path).toBe(REPORT);
|
||||
expect(files[0].dir).toBe(".ai/reports/");
|
||||
expect(files[0].name).toBe("2026-07-27-wrk-002-instance-lifecycle.md");
|
||||
expect(files[0].versions).toHaveLength(4);
|
||||
});
|
||||
|
||||
test("orders versions newest first using the API stage order", () => {
|
||||
const stages = [
|
||||
stage("implement_plan", "2026-07-27T20:00:00Z"),
|
||||
stage("simplify", "2026-07-27T18:43:11Z"),
|
||||
];
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("implement_plan", REPORT, 8422),
|
||||
artifact("simplify", REPORT, 13162),
|
||||
],
|
||||
stages,
|
||||
);
|
||||
|
||||
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
|
||||
"simplify",
|
||||
"implement_plan",
|
||||
]);
|
||||
expect(files[0].versions[0].size).toBe(13162);
|
||||
});
|
||||
|
||||
test.each([
|
||||
["equal", "2026-07-27T18:43:11Z", "2026-07-27T18:43:11Z"],
|
||||
["missing", null, null],
|
||||
])("preserves API order when stage timestamps are %s", (_case, firstAt, secondAt) => {
|
||||
const stages = [
|
||||
stage("implement_plan", firstAt),
|
||||
stage("simplify", secondAt),
|
||||
];
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("simplify", REPORT, 13162),
|
||||
artifact("implement_plan", REPORT, 8422),
|
||||
],
|
||||
stages,
|
||||
);
|
||||
|
||||
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
|
||||
"simplify",
|
||||
"implement_plan",
|
||||
]);
|
||||
});
|
||||
|
||||
test("reports the byte change each capture introduced, oldest capture first", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("implement_plan", REPORT, 8422),
|
||||
artifact("simplify", REPORT, 13162),
|
||||
artifact("consolidate_reviews", REPORT, 14323),
|
||||
artifact("fix_review_findings", REPORT, 17483),
|
||||
],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
// versions are newest-first, so deltas read 17483-14323, 14323-13162, ...
|
||||
expect(files[0].versions.map((v) => v.delta)).toEqual([3160, 1161, 4740, null]);
|
||||
});
|
||||
|
||||
test("drops captures from graph control nodes", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("start", ".ai/reports/pre-existing.md", 12402),
|
||||
artifact("plan", ".ai/plans/plan.md", 21749),
|
||||
],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
expect(files.map((file) => file.path)).toEqual([".ai/plans/plan.md"]);
|
||||
});
|
||||
|
||||
test("sorts files by their most recent capture", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("plan", ".ai/plans/plan.md", 21749),
|
||||
artifact("fix_review_findings", REPORT, 17483),
|
||||
artifact("simplify", ".ai/reviews/bugs.xml", 5231),
|
||||
],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
expect(files.map((file) => file.path)).toEqual([
|
||||
REPORT,
|
||||
".ai/reviews/bugs.xml",
|
||||
".ai/plans/plan.md",
|
||||
]);
|
||||
});
|
||||
|
||||
test("keeps retries of one stage as separate ordered versions", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[
|
||||
artifact("simplify", REPORT, 13162, 2),
|
||||
artifact("simplify", REPORT, 9000, 1),
|
||||
],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
expect(files[0].versions.map((v) => v.retry)).toEqual([2, 1]);
|
||||
expect(files[0].versions[0].size).toBe(13162);
|
||||
expect(files[0].versions.map((v) => v.delta)).toEqual([4162, null]);
|
||||
});
|
||||
|
||||
test("returns no files when every capture came from a control node", () => {
|
||||
const files = groupArtifactsByFile(
|
||||
[artifact("start", ".ai/reports/pre-existing.md", 12402)],
|
||||
STAGES,
|
||||
);
|
||||
|
||||
expect(files).toEqual([]);
|
||||
});
|
||||
});
|
||||
104
apps/fabro-web/app/routes/run-artifacts/group.ts
Normal file
104
apps/fabro-web/app/routes/run-artifacts/group.ts
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
|
||||
|
||||
import { isVisibleStage } from "../../data/runs";
|
||||
import type { Stage } from "../../lib/stage-sidebar";
|
||||
import { formatStageLabel } from "../../lib/stage-sidebar";
|
||||
|
||||
/** One capture of a file, written by a single stage attempt. */
|
||||
export interface ArtifactVersion {
|
||||
stageId: string;
|
||||
stageLabel: string;
|
||||
retry: number;
|
||||
size: number;
|
||||
/** Byte change this capture introduced; null for the first capture. */
|
||||
delta: number | null;
|
||||
}
|
||||
|
||||
/** One artifact path together with its capture history, newest first. */
|
||||
export interface ArtifactFile {
|
||||
path: string;
|
||||
/** Directory prefix including the trailing slash, or "" at the root. */
|
||||
dir: string;
|
||||
name: string;
|
||||
versions: readonly [ArtifactVersion, ...ArtifactVersion[]];
|
||||
}
|
||||
|
||||
export function splitArtifactPath(path: string): { dir: string; name: string } {
|
||||
const idx = path.lastIndexOf("/");
|
||||
return idx >= 0
|
||||
? { dir: path.slice(0, idx + 1), name: path.slice(idx + 1) }
|
||||
: { dir: "", name: path };
|
||||
}
|
||||
|
||||
interface StageInfo {
|
||||
label: string;
|
||||
order: number;
|
||||
}
|
||||
|
||||
/** Stage display data keyed by ID, preserving the API's event order. */
|
||||
function stageInfoById(stages: readonly Stage[]): Map<string, StageInfo> {
|
||||
const info = new Map<string, StageInfo>();
|
||||
stages.forEach((stage, order) => {
|
||||
info.set(stage.id, { label: formatStageLabel(stage), order });
|
||||
});
|
||||
return info;
|
||||
}
|
||||
|
||||
/**
|
||||
* Collapse raw `(stage, retry, path)` capture keys into one entry per file,
|
||||
* carrying the ordered history of every capture of that path.
|
||||
*
|
||||
* Captures from graph control nodes (`start`, `exit`) are dropped: those nodes
|
||||
* run no work, so anything they match is a pre-existing workspace file rather
|
||||
* than something the run produced.
|
||||
*/
|
||||
export function groupArtifactsByFile(
|
||||
entries: readonly RunArtifactEntry[],
|
||||
stages: readonly Stage[],
|
||||
): ArtifactFile[] {
|
||||
const stageInfo = stageInfoById(stages);
|
||||
const byPath = new Map<string, [ArtifactVersion, ...ArtifactVersion[]]>();
|
||||
|
||||
for (const entry of entries) {
|
||||
if (!isVisibleStage(entry.node_slug)) continue;
|
||||
|
||||
const info = stageInfo.get(entry.stage_id);
|
||||
const version: ArtifactVersion = {
|
||||
stageId: entry.stage_id,
|
||||
stageLabel: info?.label ?? entry.node_slug,
|
||||
retry: entry.retry,
|
||||
size: entry.size,
|
||||
delta: null,
|
||||
};
|
||||
const bucket = byPath.get(entry.relative_path);
|
||||
if (bucket) bucket.push(version);
|
||||
else byPath.set(entry.relative_path, [version]);
|
||||
}
|
||||
|
||||
const files: Array<{ file: ArtifactFile; order: number }> = [];
|
||||
for (const [path, versions] of byPath) {
|
||||
// Oldest first, so each version's delta is the change that capture introduced.
|
||||
versions.sort(
|
||||
(a, b) =>
|
||||
(stageInfo.get(a.stageId)?.order ?? -1) -
|
||||
(stageInfo.get(b.stageId)?.order ?? -1) ||
|
||||
a.retry - b.retry ||
|
||||
a.stageId.localeCompare(b.stageId),
|
||||
);
|
||||
versions.forEach((version, index) => {
|
||||
version.delta = index === 0 ? null : version.size - versions[index - 1].size;
|
||||
});
|
||||
|
||||
versions.reverse();
|
||||
const latest = versions[0];
|
||||
const { dir, name } = splitArtifactPath(path);
|
||||
files.push({
|
||||
file: { path, dir, name, versions },
|
||||
order: stageInfo.get(latest.stageId)?.order ?? -1,
|
||||
});
|
||||
}
|
||||
|
||||
// Most recently written file first — the page answers "what just happened?".
|
||||
files.sort((a, b) => b.order - a.order || a.file.path.localeCompare(b.file.path));
|
||||
return files.map((entry) => entry.file);
|
||||
}
|
||||
|
|
@ -2,11 +2,12 @@ import { afterEach, describe, expect, mock, test } from "bun:test";
|
|||
import TestRenderer from "react-test-renderer";
|
||||
|
||||
import type {
|
||||
BilledTokenCounts,
|
||||
RunBilling,
|
||||
StageTiming,
|
||||
} from "@qltysh/fabro-api-client";
|
||||
|
||||
import { makeBilledTokenCounts } from "../lib/test-fixtures";
|
||||
|
||||
function stageTiming(wall_time_ms = 0, inference_time_ms = 0, tool_time_ms = 0): StageTiming {
|
||||
return {
|
||||
wall_time_ms,
|
||||
|
|
@ -24,25 +25,12 @@ mock.module("../lib/queries", () => ({
|
|||
|
||||
const { default: RunBillingRoute } = await import("./run-billing");
|
||||
|
||||
function zeroBilling(overrides: Partial<BilledTokenCounts> = {}): BilledTokenCounts {
|
||||
return {
|
||||
cache_read_tokens: 0,
|
||||
cache_write_tokens: 0,
|
||||
input_tokens: 0,
|
||||
output_tokens: 0,
|
||||
reasoning_tokens: 0,
|
||||
total_tokens: 0,
|
||||
total_usd_micros: null,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function billing(overrides: Partial<RunBilling> = {}): RunBilling {
|
||||
return {
|
||||
stages: [],
|
||||
totals: {
|
||||
timing: stageTiming(),
|
||||
...zeroBilling(),
|
||||
...makeBilledTokenCounts(),
|
||||
},
|
||||
by_model: [],
|
||||
...overrides,
|
||||
|
|
@ -86,21 +74,21 @@ describe("RunBilling", () => {
|
|||
{
|
||||
stage: { id: "start", name: "start" },
|
||||
model: null,
|
||||
billing: zeroBilling(),
|
||||
billing: makeBilledTokenCounts(),
|
||||
timing: stageTiming(),
|
||||
state: "succeeded",
|
||||
},
|
||||
{
|
||||
stage: { id: "command", name: "command" },
|
||||
model: null,
|
||||
billing: zeroBilling(),
|
||||
billing: makeBilledTokenCounts(),
|
||||
timing: stageTiming(61000),
|
||||
state: "succeeded",
|
||||
},
|
||||
],
|
||||
totals: {
|
||||
timing: stageTiming(61000),
|
||||
...zeroBilling(),
|
||||
...makeBilledTokenCounts(),
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
|
@ -121,7 +109,7 @@ describe("RunBilling", () => {
|
|||
{
|
||||
stage: { id: "start", name: "start" },
|
||||
model: null,
|
||||
billing: zeroBilling(),
|
||||
billing: makeBilledTokenCounts(),
|
||||
timing: stageTiming(),
|
||||
state: "succeeded",
|
||||
},
|
||||
|
|
@ -131,7 +119,7 @@ describe("RunBilling", () => {
|
|||
provider: "anthropic",
|
||||
model_id: "claude-sonnet-4-5",
|
||||
},
|
||||
billing: zeroBilling({
|
||||
billing: makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -143,7 +131,7 @@ describe("RunBilling", () => {
|
|||
],
|
||||
totals: {
|
||||
timing: stageTiming(42000),
|
||||
...zeroBilling({
|
||||
...makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -157,7 +145,7 @@ describe("RunBilling", () => {
|
|||
model_id: "claude-sonnet-4-5",
|
||||
},
|
||||
stages: 1,
|
||||
billing: zeroBilling({
|
||||
billing: makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -204,7 +192,7 @@ describe("RunBilling", () => {
|
|||
model_id: "claude-opus-4-6",
|
||||
speed: "fast",
|
||||
},
|
||||
billing: zeroBilling({
|
||||
billing: makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -217,7 +205,7 @@ describe("RunBilling", () => {
|
|||
],
|
||||
totals: {
|
||||
timing: stageTiming(),
|
||||
...zeroBilling({
|
||||
...makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -232,7 +220,7 @@ describe("RunBilling", () => {
|
|||
speed: "fast",
|
||||
},
|
||||
stages: 1,
|
||||
billing: zeroBilling({
|
||||
billing: makeBilledTokenCounts({
|
||||
input_tokens: 1200,
|
||||
output_tokens: 300,
|
||||
total_tokens: 1500,
|
||||
|
|
@ -268,4 +256,4 @@ describe("RunBilling", () => {
|
|||
Date.now = originalNow;
|
||||
}
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -2,6 +2,11 @@ import { Fragment, useMemo } from "react";
|
|||
|
||||
import { EmptyState } from "../components/state";
|
||||
import { Tooltip } from "../components/ui";
|
||||
import {
|
||||
billableOutputTokens,
|
||||
billingTokenBuckets,
|
||||
hasBillingUsage,
|
||||
} from "../lib/billing";
|
||||
import {
|
||||
formatDurationMs,
|
||||
formatTokenCount,
|
||||
|
|
@ -11,6 +16,7 @@ import { useRunBilling } from "../lib/queries";
|
|||
import { IN_FLIGHT_STAGE_STATES } from "../lib/stage-sidebar";
|
||||
import { useTickingNow } from "../lib/time";
|
||||
import type {
|
||||
BilledTokenCounts,
|
||||
BillingModelRef,
|
||||
RunBilling,
|
||||
RunBillingStage,
|
||||
|
|
@ -39,23 +45,15 @@ function isInFlight(stage: RunBillingStage): boolean {
|
|||
|
||||
function isVisibleRow(row: MappedStageRow): boolean {
|
||||
if (row.inFlight) return true;
|
||||
return (
|
||||
(row.inputTokens ?? 0) > 0 ||
|
||||
(row.outputTokens ?? 0) > 0 ||
|
||||
(row.totalUsdMicros ?? 0) > 0
|
||||
);
|
||||
return row.billing != null && hasBillingUsage(row.billing);
|
||||
}
|
||||
|
||||
interface MappedStageRow {
|
||||
stage: string;
|
||||
model: string | null;
|
||||
inputTokens: number | null;
|
||||
outputTokens: number | null;
|
||||
cacheReadTokens: number | null;
|
||||
cacheWriteTokens: number | null;
|
||||
wallTimeMs: number;
|
||||
totalUsdMicros: number | null | undefined;
|
||||
inFlight: boolean;
|
||||
stage: string;
|
||||
model: string | null;
|
||||
billing: BilledTokenCounts | null;
|
||||
wallTimeMs: number;
|
||||
inFlight: boolean;
|
||||
}
|
||||
|
||||
function liveWallTimeMs(stage: RunBillingStage, now: number): number {
|
||||
|
|
@ -73,49 +71,28 @@ export const handle = { wide: true };
|
|||
function mapStageRow(stage: RunBillingStage, wallTimeMs: number): MappedStageRow {
|
||||
const hasModel = stage.model != null;
|
||||
return {
|
||||
stage: stage.stage.name,
|
||||
model: formatModelRef(stage.model),
|
||||
inputTokens: hasModel ? stage.billing.input_tokens : null,
|
||||
outputTokens: hasModel
|
||||
? stage.billing.output_tokens + stage.billing.reasoning_tokens
|
||||
: null,
|
||||
cacheReadTokens: hasModel ? stage.billing.cache_read_tokens : null,
|
||||
cacheWriteTokens: hasModel ? stage.billing.cache_write_tokens : null,
|
||||
stage: stage.stage.name,
|
||||
model: formatModelRef(stage.model),
|
||||
billing: hasModel ? stage.billing : null,
|
||||
wallTimeMs,
|
||||
totalUsdMicros: stage.billing.total_usd_micros,
|
||||
inFlight: isInFlight(stage),
|
||||
inFlight: isInFlight(stage),
|
||||
};
|
||||
}
|
||||
|
||||
/** Hover breakdown of the disjoint token buckets behind an `in / out` count. */
|
||||
function TokenBreakdown({
|
||||
cacheReadTokens,
|
||||
cacheWriteTokens,
|
||||
inputTokens,
|
||||
outputTokens,
|
||||
}: {
|
||||
cacheReadTokens: number;
|
||||
cacheWriteTokens: number;
|
||||
inputTokens: number;
|
||||
outputTokens: number;
|
||||
}) {
|
||||
const rows = [
|
||||
{ label: "Cache read", value: cacheReadTokens },
|
||||
{ label: "Cache creation", value: cacheWriteTokens },
|
||||
{ label: "Uncached", value: inputTokens },
|
||||
{ label: "Output", value: outputTokens },
|
||||
];
|
||||
function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
|
||||
const buckets = billingTokenBuckets(billing);
|
||||
return (
|
||||
<div className="min-w-44 py-0.5">
|
||||
<div className="mb-1.5 border-b border-line pb-1 font-medium text-fg-2">
|
||||
<div className="border-line text-fg-2 mb-1.5 border-b pb-1 font-medium">
|
||||
Tokens in / out
|
||||
</div>
|
||||
<dl className="grid grid-cols-[1fr_auto] gap-x-6 gap-y-1">
|
||||
{rows.map((row) => (
|
||||
<Fragment key={row.label}>
|
||||
<dt className="text-fg-3">{row.label}</dt>
|
||||
<dd className="text-right font-mono tabular-nums text-fg">
|
||||
{formatTokens(row.value)}
|
||||
{buckets.map((bucket) => (
|
||||
<Fragment key={bucket.label}>
|
||||
<dt className="text-fg-3">{bucket.label}</dt>
|
||||
<dd className="text-fg text-right font-mono tabular-nums">
|
||||
{formatTokens(bucket.value)}
|
||||
</dd>
|
||||
</Fragment>
|
||||
))}
|
||||
|
|
@ -128,42 +105,16 @@ function TokenBreakdown({
|
|||
* Renders an `input / output` token count. When the row has model usage,
|
||||
* hovering the count reveals the cache breakdown.
|
||||
*/
|
||||
function TokensCell({
|
||||
inputTokens,
|
||||
outputTokens,
|
||||
cacheReadTokens,
|
||||
cacheWriteTokens,
|
||||
}: {
|
||||
inputTokens: number | null;
|
||||
outputTokens: number | null;
|
||||
cacheReadTokens: number | null;
|
||||
cacheWriteTokens: number | null;
|
||||
}) {
|
||||
function TokensCell({ billing }: { billing: BilledTokenCounts | null }) {
|
||||
const display = (
|
||||
<>
|
||||
{formatTokens(inputTokens)} <span className="text-fg-muted">/</span>{" "}
|
||||
{formatTokens(outputTokens)}
|
||||
{formatTokens(billing?.input_tokens)} <span className="text-fg-muted">/</span>{" "}
|
||||
{formatTokens(billing ? billableOutputTokens(billing) : null)}
|
||||
</>
|
||||
);
|
||||
if (
|
||||
inputTokens == null ||
|
||||
outputTokens == null ||
|
||||
cacheReadTokens == null ||
|
||||
cacheWriteTokens == null
|
||||
) {
|
||||
return display;
|
||||
}
|
||||
if (!billing) return display;
|
||||
return (
|
||||
<Tooltip
|
||||
label={
|
||||
<TokenBreakdown
|
||||
cacheReadTokens={cacheReadTokens}
|
||||
cacheWriteTokens={cacheWriteTokens}
|
||||
inputTokens={inputTokens}
|
||||
outputTokens={outputTokens}
|
||||
/>
|
||||
}
|
||||
>
|
||||
<Tooltip label={<TokenBreakdown billing={billing} />}>
|
||||
<span>{display}</span>
|
||||
</Tooltip>
|
||||
);
|
||||
|
|
@ -189,15 +140,14 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
if (!billing) return [];
|
||||
return billing.by_model
|
||||
.map((entry) => ({
|
||||
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
|
||||
stages: entry.stages,
|
||||
inputTokens: entry.billing.input_tokens,
|
||||
outputTokens: entry.billing.output_tokens + entry.billing.reasoning_tokens,
|
||||
cacheReadTokens: entry.billing.cache_read_tokens,
|
||||
cacheWriteTokens: entry.billing.cache_write_tokens,
|
||||
totalUsdMicros: entry.billing.total_usd_micros,
|
||||
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
|
||||
stages: entry.stages,
|
||||
billing: entry.billing,
|
||||
}))
|
||||
.sort((a, b) => (b.totalUsdMicros ?? -1) - (a.totalUsdMicros ?? -1));
|
||||
.sort(
|
||||
(a, b) =>
|
||||
(b.billing.total_usd_micros ?? -1) - (a.billing.total_usd_micros ?? -1),
|
||||
);
|
||||
}, [billing]);
|
||||
|
||||
// Re-derive only the in-flight rows on each tick; everything else stays put.
|
||||
|
|
@ -218,16 +168,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
: (billing?.totals.timing.wall_time_ms ?? 0);
|
||||
|
||||
const hasLlmStages = (billing?.by_model.length ?? 0) > 0;
|
||||
const totalInput = hasLlmStages ? (billing?.totals.input_tokens ?? null) : null;
|
||||
const totalOutput = hasLlmStages && billing
|
||||
? billing.totals.output_tokens + billing.totals.reasoning_tokens
|
||||
: null;
|
||||
const totalCacheRead = hasLlmStages
|
||||
? (billing?.totals.cache_read_tokens ?? null)
|
||||
: null;
|
||||
const totalCacheWrite = hasLlmStages
|
||||
? (billing?.totals.cache_write_tokens ?? null)
|
||||
: null;
|
||||
const totalBilling = hasLlmStages && billing ? billing.totals : null;
|
||||
const totalUsdMicros = billing?.totals.total_usd_micros;
|
||||
const modelStageCount = modelBreakdown.reduce((sum, row) => sum + row.stages, 0);
|
||||
const visibleRows = rows.filter(isVisibleRow);
|
||||
|
|
@ -268,18 +209,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
{row.model ?? EMPTY_VALUE}
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
|
||||
<TokensCell
|
||||
inputTokens={row.inputTokens}
|
||||
outputTokens={row.outputTokens}
|
||||
cacheReadTokens={row.cacheReadTokens}
|
||||
cacheWriteTokens={row.cacheWriteTokens}
|
||||
/>
|
||||
<TokensCell billing={row.billing} />
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
|
||||
{formatDurationMs(row.wallTimeMs)}
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
|
||||
{formatUsdMicrosOrDash(row.totalUsdMicros)}
|
||||
{formatUsdMicrosOrDash(row.billing?.total_usd_micros)}
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
|
|
@ -289,12 +225,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
<td className="px-4 py-3 font-medium text-fg">Total</td>
|
||||
<td className="px-4 py-3 text-xs text-fg-muted">All models</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
|
||||
<TokensCell
|
||||
inputTokens={totalInput}
|
||||
outputTokens={totalOutput}
|
||||
cacheReadTokens={totalCacheRead}
|
||||
cacheWriteTokens={totalCacheWrite}
|
||||
/>
|
||||
<TokensCell billing={totalBilling} />
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
|
||||
{formatDurationMs(totalWallTimeMs)}
|
||||
|
|
@ -328,15 +259,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
{row.stages}
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
|
||||
<TokensCell
|
||||
inputTokens={row.inputTokens}
|
||||
outputTokens={row.outputTokens}
|
||||
cacheReadTokens={row.cacheReadTokens}
|
||||
cacheWriteTokens={row.cacheWriteTokens}
|
||||
/>
|
||||
<TokensCell billing={row.billing} />
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
|
||||
{formatUsdMicrosOrDash(row.totalUsdMicros)}
|
||||
{formatUsdMicrosOrDash(row.billing.total_usd_micros)}
|
||||
</td>
|
||||
</tr>
|
||||
))}
|
||||
|
|
@ -348,12 +274,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
|
|||
{modelStageCount}
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
|
||||
<TokensCell
|
||||
inputTokens={totalInput}
|
||||
outputTokens={totalOutput}
|
||||
cacheReadTokens={totalCacheRead}
|
||||
cacheWriteTokens={totalCacheWrite}
|
||||
/>
|
||||
<TokensCell billing={totalBilling} />
|
||||
</td>
|
||||
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
|
||||
{formatUsdMicrosOrDash(totalUsdMicros)}
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ import {
|
|||
} from "../components/ui";
|
||||
import { mutateRunListCaches } from "../lib/board-cache";
|
||||
import { useDemoMode } from "../lib/demo-mode";
|
||||
import { useTickingNow } from "../lib/time";
|
||||
import { useSWRConfig } from "swr";
|
||||
import {
|
||||
useArchiveRun,
|
||||
|
|
@ -56,10 +57,7 @@ import {
|
|||
lifecycleActionVisibility,
|
||||
updateLifecycleToastState,
|
||||
} from "./run-detail/lifecycle-toasts";
|
||||
import {
|
||||
buildRunDetailRun,
|
||||
useTickingNow,
|
||||
} from "./run-detail/model";
|
||||
import { buildRunDetailRun } from "./run-detail/model";
|
||||
import {
|
||||
buildRunDetailTabs,
|
||||
childRouteLayoutFlags,
|
||||
|
|
@ -69,6 +67,8 @@ import {
|
|||
|
||||
export const handle = { hideHeader: true };
|
||||
|
||||
const RUN_TIMING_REFRESH_INTERVAL_MS = 30_000;
|
||||
|
||||
type LifecycleTrigger = () => Promise<LifecycleMutationResult | undefined>;
|
||||
|
||||
export interface DockMeasurement {
|
||||
|
|
@ -95,7 +95,7 @@ export function meta({ data }: any) {
|
|||
|
||||
export default function RunDetail({ params }: { params: { id: string } }) {
|
||||
const demoMode = useDemoMode();
|
||||
const runQuery = useRun(params.id);
|
||||
const runQuery = useRun(params.id, RUN_TIMING_REFRESH_INTERVAL_MS);
|
||||
const runStateQuery = useRunState(params.id);
|
||||
const summary = runQuery.data;
|
||||
const run = summary ? buildRunDetailRun(summary) : null;
|
||||
|
|
@ -135,7 +135,10 @@ export default function RunDetail({ params }: { params: { id: string } }) {
|
|||
childrenCount,
|
||||
});
|
||||
const steerBarRef = useRef<SteerBarHandle | null>(null);
|
||||
const now = useTickingNow(30_000);
|
||||
const now = useTickingNow(
|
||||
summary != null && summary.timestamps.completed_at == null,
|
||||
RUN_TIMING_REFRESH_INTERVAL_MS,
|
||||
);
|
||||
const { fullHeight, hideSteerBar } = childRouteLayoutFlags(matches);
|
||||
|
||||
useRunEvents(params.id);
|
||||
|
|
|
|||
|
|
@ -327,6 +327,7 @@ function DurationPopover({
|
|||
}) {
|
||||
const endMs = completedAt != null ? Date.parse(completedAt) : now;
|
||||
const sinceCreatedMs = Math.max(0, endMs - Date.parse(createdAt));
|
||||
const isRunning = completedAt == null;
|
||||
return (
|
||||
<>
|
||||
<PopoverHeader>Duration</PopoverHeader>
|
||||
|
|
@ -336,8 +337,14 @@ function DurationPopover({
|
|||
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(sinceCreatedMs)}</dd>
|
||||
</div>
|
||||
<div>
|
||||
<dt className="text-fg-3">Active (inference + tools)</dt>
|
||||
<dt className="text-fg-3">
|
||||
Active (inference + tools){isRunning ? " — estimated" : ""}
|
||||
</dt>
|
||||
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(timing.active_time_ms)}</dd>
|
||||
<dd className="mt-0.5 text-fg-3">
|
||||
{formatDurationMs(timing.inference_time_ms)} inference ·{" "}
|
||||
{formatDurationMs(timing.tool_time_ms)} tools
|
||||
</dd>
|
||||
</div>
|
||||
</dl>
|
||||
</>
|
||||
|
|
|
|||
|
|
@ -1,6 +1,3 @@
|
|||
import { useState } from "react";
|
||||
|
||||
import { useInterval } from "../../hooks/effects";
|
||||
import {
|
||||
isRunStatus,
|
||||
mapRunToRunItem,
|
||||
|
|
@ -8,12 +5,6 @@ import {
|
|||
type Run,
|
||||
} from "../../data/runs";
|
||||
|
||||
export function useTickingNow(intervalMs: number): number {
|
||||
const [now, setNow] = useState(() => Date.now());
|
||||
useInterval(() => setNow(Date.now()), intervalMs);
|
||||
return now;
|
||||
}
|
||||
|
||||
export type RunDetailRun = ReturnType<typeof mapRunToRunItem> & {
|
||||
statusLabel: string;
|
||||
statusDot: string;
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import { renderToStaticMarkup } from "react-dom/server";
|
|||
import { StageState } from "@qltysh/fabro-api-client";
|
||||
|
||||
import type { Stage } from "../lib/stage-sidebar";
|
||||
import { makeBilledTokenCounts } from "../lib/test-fixtures";
|
||||
import { StageChatView } from "./run-stages";
|
||||
|
||||
function stage(overrides: Partial<Stage> = {}): Stage {
|
||||
|
|
@ -18,6 +19,7 @@ function stage(overrides: Partial<Stage> = {}): Stage {
|
|||
resumedFromStageId: null,
|
||||
startedAt: "2026-04-09T12:00:00Z",
|
||||
providerUsed: null,
|
||||
billing: makeBilledTokenCounts(),
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,9 +1,14 @@
|
|||
import { describe, expect, test } from "bun:test";
|
||||
import { renderToStaticMarkup } from "react-dom/server";
|
||||
|
||||
import type { ReasoningOutput } from "@qltysh/fabro-api-client";
|
||||
import type {
|
||||
BilledTokenCounts,
|
||||
ReasoningOutput,
|
||||
StageModelUsage,
|
||||
} from "@qltysh/fabro-api-client";
|
||||
|
||||
import { EventDetails } from "./run-stages";
|
||||
import { makeBilledTokenCounts } from "../lib/test-fixtures";
|
||||
import { EventDetails, ModelUsagePopover } from "./run-stages";
|
||||
|
||||
const RUN_START = "2026-04-09T12:00:00Z";
|
||||
|
||||
|
|
@ -72,3 +77,77 @@ describe("EventDetails reasoning", () => {
|
|||
expect(html).toContain(`${"x".repeat(280)}…`);
|
||||
});
|
||||
});
|
||||
|
||||
const PROVIDER_USED: StageModelUsage = {
|
||||
mode: "agent",
|
||||
provider: "moonshot",
|
||||
model: "kimi-k3",
|
||||
reasoning_effort: "max",
|
||||
};
|
||||
|
||||
function popoverMarkup(counts: BilledTokenCounts): string {
|
||||
return renderToStaticMarkup(
|
||||
<ModelUsagePopover providerUsed={PROVIDER_USED} billing={counts} />,
|
||||
);
|
||||
}
|
||||
|
||||
describe("ModelUsagePopover billing", () => {
|
||||
test("shows the visit's token buckets and cost next to the model", () => {
|
||||
const html = popoverMarkup(
|
||||
makeBilledTokenCounts({
|
||||
input_tokens: 28_640,
|
||||
output_tokens: 7_550,
|
||||
reasoning_tokens: 1_200,
|
||||
cache_read_tokens: 4_800,
|
||||
cache_write_tokens: 1_500,
|
||||
total_tokens: 43_690,
|
||||
total_usd_micros: 720_000,
|
||||
}),
|
||||
);
|
||||
|
||||
expect(html).toContain("kimi-k3");
|
||||
expect(html).toContain("Cache read");
|
||||
expect(html).toContain("4.8k");
|
||||
expect(html).toContain("Cache creation");
|
||||
expect(html).toContain("1.5k");
|
||||
expect(html).toContain("Uncached");
|
||||
expect(html).toContain("28.6k");
|
||||
// Output folds in reasoning tokens, matching the Billing tab.
|
||||
expect(html).toContain("Output");
|
||||
expect(html).toContain("8.8k");
|
||||
expect(html).toContain("Cost");
|
||||
expect(html).toContain("$0.72");
|
||||
});
|
||||
|
||||
test("omits the token section for a stage that called no model", () => {
|
||||
const html = popoverMarkup(makeBilledTokenCounts());
|
||||
|
||||
expect(html).toContain("kimi-k3");
|
||||
expect(html).not.toContain("Tokens");
|
||||
expect(html).not.toContain("Cost");
|
||||
});
|
||||
|
||||
test("still shows tokens when nothing priced the stage", () => {
|
||||
const html = popoverMarkup(
|
||||
makeBilledTokenCounts({
|
||||
input_tokens: 1_000,
|
||||
output_tokens: 500,
|
||||
total_tokens: 1_500,
|
||||
}),
|
||||
);
|
||||
|
||||
expect(html).toContain("Uncached");
|
||||
expect(html).toContain("1.0k");
|
||||
expect(html).not.toContain("Cost");
|
||||
});
|
||||
|
||||
test("shows a provider-reported cost when token counts are unavailable", () => {
|
||||
const html = popoverMarkup(
|
||||
makeBilledTokenCounts({ total_usd_micros: 720_000 }),
|
||||
);
|
||||
|
||||
expect(html).toContain("kimi-k3");
|
||||
expect(html).toContain("Cost");
|
||||
expect(html).toContain("$0.72");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -67,7 +67,9 @@ import {
|
|||
formatBytes,
|
||||
formatDurationMs,
|
||||
formatTokenCount,
|
||||
formatUsdMicros,
|
||||
} from "../lib/format";
|
||||
import { billingTokenBuckets, hasBillingUsage } from "../lib/billing";
|
||||
import { plural } from "../lib/plural";
|
||||
import {
|
||||
useRun,
|
||||
|
|
@ -93,6 +95,7 @@ import {
|
|||
type UnknownRecord,
|
||||
} from "../lib/unknown";
|
||||
import type {
|
||||
BilledTokenCounts,
|
||||
EventEnvelope,
|
||||
ReasoningOutput,
|
||||
StageHandler,
|
||||
|
|
@ -866,10 +869,42 @@ export function formatStageModelUsageLabel(
|
|||
return effort ? `${model}[${effort}]` : model;
|
||||
}
|
||||
|
||||
function ModelUsagePopover({
|
||||
const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
|
||||
|
||||
/** Tokens and cost for this stage visit alone. */
|
||||
function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
|
||||
if (!hasBillingUsage(billing)) return null;
|
||||
const buckets = billingTokenBuckets(billing);
|
||||
const cost = formatUsdMicros(billing.total_usd_micros);
|
||||
return (
|
||||
<div className="mt-3">
|
||||
<PopoverHeader>Tokens</PopoverHeader>
|
||||
<PopoverRows>
|
||||
{buckets.map((bucket) => (
|
||||
<PopoverRow key={bucket.label} label={bucket.label}>
|
||||
<span className={POPOVER_NUMBER}>
|
||||
{bucket.value === 0
|
||||
? "0"
|
||||
: formatTokenCount(bucket.value, { compactDecimal: true })}
|
||||
</span>
|
||||
</PopoverRow>
|
||||
))}
|
||||
{cost && (
|
||||
<PopoverRow label="Cost">
|
||||
<span className={POPOVER_NUMBER}>{cost}</span>
|
||||
</PopoverRow>
|
||||
)}
|
||||
</PopoverRows>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export function ModelUsagePopover({
|
||||
providerUsed,
|
||||
billing,
|
||||
}: {
|
||||
providerUsed: StageModelUsage;
|
||||
billing: BilledTokenCounts;
|
||||
}) {
|
||||
return (
|
||||
<>
|
||||
|
|
@ -892,6 +927,7 @@ function ModelUsagePopover({
|
|||
<PopoverRow label="Speed">{providerUsed.speed}</PopoverRow>
|
||||
)}
|
||||
</PopoverRows>
|
||||
<StageBillingRows billing={billing} />
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
|
@ -1905,6 +1941,7 @@ function EventsToolbar({
|
|||
filteredCount,
|
||||
totalCount,
|
||||
providerUsed,
|
||||
billing,
|
||||
events,
|
||||
runId,
|
||||
stageId,
|
||||
|
|
@ -1924,6 +1961,7 @@ function EventsToolbar({
|
|||
filteredCount: number;
|
||||
totalCount: number;
|
||||
providerUsed: StageModelUsage | null;
|
||||
billing: BilledTokenCounts;
|
||||
events: EventEnvelope[];
|
||||
runId: string;
|
||||
stageId: string;
|
||||
|
|
@ -2004,7 +2042,9 @@ function EventsToolbar({
|
|||
className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${
|
||||
showFilters ? "" : "ml-auto"
|
||||
}`}
|
||||
content={<ModelUsagePopover providerUsed={providerUsed} />}
|
||||
content={
|
||||
<ModelUsagePopover providerUsed={providerUsed} billing={billing} />
|
||||
}
|
||||
>
|
||||
<CpuChipIcon className="size-3.5" aria-hidden="true" />
|
||||
<span className="font-mono">{modelUsageLabel}</span>
|
||||
|
|
@ -2359,6 +2399,7 @@ function RunStageActivityStage({
|
|||
effectiveTab === "primary" ? turns.length : debugEvents.length
|
||||
}
|
||||
providerUsed={selectedStage.providerUsed}
|
||||
billing={selectedStage.billing}
|
||||
events={stageEventsQuery.data ?? []}
|
||||
runId={runId}
|
||||
stageId={selectedStageId}
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ import { ciConfig, columnForRun, columnStatusDisplay, columnStatuses, deriveCiSt
|
|||
import type { CiStatus, CheckRun, CheckStatus, RunItem } from "../data/runs";
|
||||
import { EmptyState } from "../components/state";
|
||||
import { PullRequestChip } from "../components/pull-request-chip";
|
||||
import { SizeChip } from "../components/size-chip";
|
||||
import {
|
||||
summarizeBatchLifecycleAction,
|
||||
} from "../components/runs-list/batch-lifecycle";
|
||||
|
|
@ -345,7 +346,7 @@ function PrCard({
|
|||
|
||||
// All inline footer metadata on PrCard belongs in this one row. Adding a new
|
||||
// piece as a sibling `<div>` below the card body recreates a recurring bug
|
||||
// where stats stack onto separate lines instead of sitting next to elapsed/actions.
|
||||
// where stats stack onto separate lines instead of sitting next to size/actions.
|
||||
function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
|
||||
const hasActions = actions != null && actions.length > 0;
|
||||
const hasStats =
|
||||
|
|
@ -354,7 +355,7 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
|
|||
(pr.additions != null && pr.additions !== 0) ||
|
||||
(pr.deletions != null && pr.deletions !== 0);
|
||||
|
||||
if (!hasStats && !hasActions && pr.elapsed == null) return null;
|
||||
if (!hasStats && !hasActions && pr.size == null) return null;
|
||||
|
||||
return (
|
||||
<div className="mt-3 flex items-center gap-3 font-mono text-xs">
|
||||
|
|
@ -416,9 +417,9 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
|
|||
))}
|
||||
</div>
|
||||
)}
|
||||
{pr.elapsed != null && (
|
||||
<span className={`text-fg-muted ${hasActions ? "" : "ml-auto"}`}>
|
||||
{pr.elapsed}
|
||||
{pr.size != null && (
|
||||
<span className={hasActions ? "inline-flex" : "ml-auto inline-flex"}>
|
||||
<SizeChip size={pr.size} totalUsdMicros={pr.totalUsdMicros} />
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
|
|
|
|||
|
|
@ -155,9 +155,16 @@ git diff --check
|
|||
|
||||
- Inference time is Fabro-observed LLM request/stream elapsed time, not
|
||||
provider-reported model-only compute time.
|
||||
- LLM retry backoff, queueing outside a request/stream, human waits, steering
|
||||
waits, and scheduler gaps are wall time but not active time.
|
||||
- Active timing is finalized-event based in v1; live active-time ticking can be
|
||||
added later if it becomes necessary.
|
||||
- Queueing outside a request/stream, human waits, steering waits, and scheduler
|
||||
gaps are wall time but not active time. Retry delay inside an open LLM request
|
||||
bracket follows the executor stopwatch and counts as inference time.
|
||||
- ~~Active timing is finalized-event based in v1; live active-time ticking can
|
||||
be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became
|
||||
necessary: a run parked in one long agent stage reported ~12% of its wall time
|
||||
as active, because in-flight stages contributed nothing. Stage projections now
|
||||
accumulate inference and tool brackets from the event log and expose
|
||||
`StageProjection::live_timing(now)`, the active-time twin of
|
||||
`live_wall_time_ms`. Finalized values remain authoritative and still replace
|
||||
the live estimate at terminal events. Implemented in PR #647.
|
||||
- No compatibility layer is required for existing API clients or stored run
|
||||
event data.
|
||||
|
|
|
|||
|
|
@ -10673,7 +10673,38 @@ components:
|
|||
- type: "null"
|
||||
description: |
|
||||
Per-attempt timing breakdown for the latest terminal attempt:
|
||||
wall time plus the active inference/tool breakdown.
|
||||
wall time plus the active inference/tool breakdown. Null while the
|
||||
stage is still in flight; the live estimate is derived from
|
||||
`live_inference_ms`, `live_tool_ms`, and any open bracket.
|
||||
live_inference_ms:
|
||||
type: integer
|
||||
format: uint64
|
||||
minimum: 0
|
||||
default: 0
|
||||
description: |
|
||||
Inference time accumulated from closed brackets during the current
|
||||
attempt. Live estimate only — the authoritative value arrives with
|
||||
the terminal event and lands in `timing`. Excludes the currently
|
||||
open bracket, whose span is measured from `inference.started_at`.
|
||||
example: 78230
|
||||
live_tool_ms:
|
||||
type: integer
|
||||
format: uint64
|
||||
minimum: 0
|
||||
default: 0
|
||||
description: |
|
||||
Tool time accumulated from closed tool batches during the current
|
||||
attempt. A batch spans the first dispatched call through the
|
||||
completion that drains the last outstanding one, so tools running
|
||||
concurrently within a turn are counted once.
|
||||
example: 7588
|
||||
tool_batch:
|
||||
oneOf:
|
||||
- $ref: "#/components/schemas/StageToolBatchProjection"
|
||||
- type: "null"
|
||||
description: |
|
||||
Open tool batch: when the batch started and which calls have not
|
||||
yet reported completion.
|
||||
usage:
|
||||
$ref: "#/components/schemas/BilledTokenCounts"
|
||||
model:
|
||||
|
|
@ -10727,6 +10758,13 @@ components:
|
|||
Open inference bracket, if the event log contains one. Present means
|
||||
a model request was dispatched and no closing event has been seen —
|
||||
not that the model is computing right now.
|
||||
acp_started_at:
|
||||
type: ["string", "null"]
|
||||
format: date-time
|
||||
description: >
|
||||
Start of an external ACP agent process, if one is running. ACP
|
||||
agents do not expose Fabro's internal LLM brackets, so the process
|
||||
lifetime supplies their live inference estimate.
|
||||
agent_control:
|
||||
$ref: "#/components/schemas/AgentControlState"
|
||||
description: Whether the agent is executing normally or waiting for steering after an interrupt.
|
||||
|
|
@ -10734,6 +10772,37 @@ components:
|
|||
$ref: "#/components/schemas/StageState"
|
||||
description: Lifecycle state of the stage projection.
|
||||
|
||||
StageToolBatchProjection:
|
||||
description: >
|
||||
One open tool batch: tool calls dispatched together that have not all
|
||||
reported completion. `open_call_ids` is a set rather than a count so a
|
||||
duplicated completion in a replayed log cannot drain the batch early.
|
||||
type: object
|
||||
required:
|
||||
- session_id
|
||||
- started_at
|
||||
- open_call_ids
|
||||
properties:
|
||||
session_id:
|
||||
type: string
|
||||
description: >
|
||||
Root agent session that dispatched the batch. Transitions are
|
||||
gated on it so delayed events from a replaced session cannot
|
||||
mutate the current batch.
|
||||
started_at:
|
||||
type: string
|
||||
format: date-time
|
||||
description: >
|
||||
When the batch opened — the first dispatched call observed while no
|
||||
other calls were outstanding.
|
||||
open_call_ids:
|
||||
type: array
|
||||
minItems: 1
|
||||
uniqueItems: true
|
||||
items:
|
||||
type: string
|
||||
description: Calls dispatched but not yet completed, by tool call id.
|
||||
|
||||
StageInferenceProjection:
|
||||
description: >
|
||||
One open inference bracket: a dispatched LLM request that has not yet
|
||||
|
|
@ -12244,9 +12313,17 @@ components:
|
|||
observed LLM request/stream elapsed time; `tool_time_ms` is tool or
|
||||
command execution elapsed time; `active_time_ms` equals
|
||||
`inference_time_ms + tool_time_ms`.
|
||||
|
||||
For a terminal stage these come from the worker's own stopwatch and are
|
||||
authoritative. For a stage still in flight they are a live estimate
|
||||
reconstructed from the event log, and `active_time_ms` is clamped to
|
||||
`wall_time_ms`. The estimate is replaced by the authoritative
|
||||
breakdown when the stage reaches a terminal event.
|
||||
type: object
|
||||
required:
|
||||
- wall_time_ms
|
||||
- inference_time_ms
|
||||
- tool_time_ms
|
||||
- active_time_ms
|
||||
properties:
|
||||
wall_time_ms:
|
||||
|
|
@ -12278,9 +12355,16 @@ components:
|
|||
Timing rollup for an entire run. Active fields sum work across stage
|
||||
visits, so `active_time_ms` can exceed `wall_time_ms` when parallel
|
||||
branches run concurrently.
|
||||
|
||||
For a running run, stages still in flight contribute a live estimate
|
||||
rather than nothing, so wall and active both advance continuously.
|
||||
Unlike `StageTiming`, active is not clamped to wall here — concurrent
|
||||
branches can legitimately sum past run wall time.
|
||||
type: object
|
||||
required:
|
||||
- wall_time_ms
|
||||
- inference_time_ms
|
||||
- tool_time_ms
|
||||
- active_time_ms
|
||||
properties:
|
||||
wall_time_ms:
|
||||
|
|
@ -12609,6 +12693,7 @@ components:
|
|||
- status
|
||||
- node_id
|
||||
- visit
|
||||
- billing
|
||||
properties:
|
||||
id:
|
||||
$ref: "#/components/schemas/StageId"
|
||||
|
|
@ -12668,6 +12753,15 @@ components:
|
|||
format: date-time
|
||||
description: Wall-clock time the latest attempt of this stage started, if known.
|
||||
example: "2026-04-29T12:34:56Z"
|
||||
billing:
|
||||
$ref: "#/components/schemas/BilledTokenCounts"
|
||||
description: >-
|
||||
Token counts for this stage execution alone. `total_usd_micros` is
|
||||
the provider-reported cost when there is one, otherwise the server
|
||||
catalog's price for these tokens — the same pricing the
|
||||
`/runs/{id}/billing` rows use. All-zero counts mean the stage made
|
||||
no model calls. Unlike the billing rows, which sum every visit of a
|
||||
node, this covers only this visit.
|
||||
|
||||
# ── File Diff Schemas ──────────────────────────────────────────────
|
||||
|
||||
|
|
|
|||
|
|
@ -113,7 +113,7 @@ plan [label="Plan", prompt="Create an implementation plan."]
|
|||
|
||||
**Node identifiers** must start with a letter or underscore, followed by letters, digits, or underscores (e.g. `run_tests`, `gate_1`, `_private`).
|
||||
|
||||
Nodes referenced in edges are auto-created if not explicitly declared.
|
||||
Every node used by an edge needs its own declaration. Validation fails when an edge names a node the workflow never declares, because that is nearly always a typo or a rename that missed an edge. The declaration can come before or after the edges that use it, and it can live in a subgraph.
|
||||
|
||||
### Edge declarations
|
||||
|
||||
|
|
|
|||
|
|
@ -61,11 +61,8 @@ pub(crate) async fn create_run(
|
|||
None
|
||||
};
|
||||
|
||||
let mut validation = manifest_validation::validate_manifest(
|
||||
&RunLayer::default(),
|
||||
&built.manifest,
|
||||
ctx.catalog()?,
|
||||
)?;
|
||||
let mut validation =
|
||||
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
|
||||
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
|
||||
let diagnostics = api_diagnostics_to_local(&validation.workflow.diagnostics);
|
||||
if !quiet {
|
||||
|
|
|
|||
|
|
@ -110,7 +110,6 @@ pub(crate) async fn execute(
|
|||
run_id,
|
||||
run_spec.source_directory.as_deref(),
|
||||
&run_dir,
|
||||
Arc::clone(&catalog),
|
||||
)
|
||||
} else {
|
||||
None
|
||||
|
|
@ -237,13 +236,12 @@ fn build_fabro_run_tool_services(
|
|||
current_run_id: RunId,
|
||||
source_directory: Option<&str>,
|
||||
run_dir: &Path,
|
||||
catalog: Arc<Catalog>,
|
||||
) -> Option<FabroRunToolServices> {
|
||||
if worker_token.trim().is_empty() {
|
||||
return None;
|
||||
}
|
||||
let backend = ClientBackend::new(Arc::new(client))
|
||||
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder { catalog }));
|
||||
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder));
|
||||
Some(FabroRunToolServices {
|
||||
backend: Arc::new(backend),
|
||||
current_run_id,
|
||||
|
|
@ -252,9 +250,7 @@ fn build_fabro_run_tool_services(
|
|||
})
|
||||
}
|
||||
|
||||
struct WorkerRunManifestBuilder {
|
||||
catalog: Arc<Catalog>,
|
||||
}
|
||||
struct WorkerRunManifestBuilder;
|
||||
|
||||
impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
|
||||
fn build_run_manifest(
|
||||
|
|
@ -263,12 +259,7 @@ impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
|
|||
cwd: &Path,
|
||||
user_settings_path: &Path,
|
||||
) -> fabro_tool::ToolResult<RunManifest> {
|
||||
run_tool_manifest::build_run_tool_manifest(
|
||||
spec,
|
||||
cwd,
|
||||
user_settings_path,
|
||||
Arc::clone(&self.catalog),
|
||||
)
|
||||
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -23,11 +23,7 @@ pub(crate) fn run(
|
|||
user_settings_path: Some(active_settings_path(None)),
|
||||
..Default::default()
|
||||
})?;
|
||||
let response = manifest_validation::validate_manifest(
|
||||
&RunLayer::default(),
|
||||
&built.manifest,
|
||||
base_ctx.catalog()?,
|
||||
)?;
|
||||
let response = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
|
||||
let diagnostics = api_diagnostics_to_local(&response.workflow.diagnostics);
|
||||
|
||||
if base_ctx.json_output() {
|
||||
|
|
|
|||
|
|
@ -49,54 +49,49 @@ where
|
|||
|
||||
pub(crate) fn print_diagnostics(diagnostics: &[Diagnostic], styles: &Styles, printer: Printer) {
|
||||
for d in diagnostics {
|
||||
let location = match (&d.node_id, &d.edge) {
|
||||
(Some(node), _) => format!(" [node: {node}]"),
|
||||
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
|
||||
_ => String::new(),
|
||||
};
|
||||
let source_prefix = source_prefix(d);
|
||||
match d.severity {
|
||||
Severity::Error if source_prefix.is_empty() => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}{location}: {} ({})",
|
||||
styles.red.apply_to("error"),
|
||||
d.message,
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Error => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}: {source_prefix}{}{location} ({})",
|
||||
styles.red.apply_to("error"),
|
||||
d.message,
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}{location}: {} ({})",
|
||||
styles.yellow.apply_to("warning"),
|
||||
d.message,
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Warning => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}: {source_prefix}{}{location} ({})",
|
||||
styles.yellow.apply_to("warning"),
|
||||
d.message,
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Info => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}",
|
||||
styles.dim.apply_to(if source_prefix.is_empty() {
|
||||
format!("info{location}: {} ({})", d.message, d.rule)
|
||||
} else {
|
||||
format!("info: {source_prefix}{}{location} ({})", d.message, d.rule)
|
||||
}),
|
||||
),
|
||||
print_diagnostic(d, styles, printer);
|
||||
// The fix is the actionable half of a diagnostic, so it follows every
|
||||
// severity rather than hiding behind --verbose. Rules that have nothing
|
||||
// useful to suggest leave it unset.
|
||||
if let Some(fix) = &d.fix {
|
||||
fabro_util::printerr!(printer, " {} {fix}", styles.dim.apply_to("fix:"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn print_diagnostic(d: &Diagnostic, styles: &Styles, printer: Printer) {
|
||||
let location = match (&d.node_id, &d.edge) {
|
||||
(Some(node), _) => format!(" [node: {node}]"),
|
||||
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
|
||||
_ => String::new(),
|
||||
};
|
||||
let source_prefix = source_prefix(d);
|
||||
let body = if source_prefix.is_empty() {
|
||||
format!("{location}: {}", d.message)
|
||||
} else {
|
||||
format!(": {source_prefix}{}{location}", d.message)
|
||||
};
|
||||
match d.severity {
|
||||
Severity::Error => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}{body} ({})",
|
||||
styles.red.apply_to("error"),
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Warning => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}{body} ({})",
|
||||
styles.yellow.apply_to("warning"),
|
||||
styles.dim.apply_to(&d.rule),
|
||||
),
|
||||
Severity::Info => fabro_util::printerr!(
|
||||
printer,
|
||||
"{}",
|
||||
styles.dim.apply_to(format!("info{body} ({})", d.rule)),
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
fn source_prefix(diagnostic: &Diagnostic) -> String {
|
||||
match (
|
||||
diagnostic.source_path.as_deref(),
|
||||
|
|
|
|||
|
|
@ -101,6 +101,40 @@ fn create_uses_explicit_server_target_and_prints_remote_run_id() {
|
|||
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn create_defers_provider_validation_to_the_server() {
|
||||
let context = test_context!();
|
||||
let server = MockServer::start();
|
||||
let run_id = unique_run_id();
|
||||
let mock = server.mock(|when, then| {
|
||||
when.method("POST")
|
||||
.path("/api/v1/runs")
|
||||
.body_includes(r#"provider=\"server-only\""#);
|
||||
then.status(201)
|
||||
.header("Content-Type", "application/json")
|
||||
.body(run_status_response(run_id.as_str(), "submitted").to_string());
|
||||
});
|
||||
let output = context
|
||||
.create_cmd()
|
||||
.args([
|
||||
"--server",
|
||||
&format!("{}/api/v1", server.base_url()),
|
||||
"--dry-run",
|
||||
fixture("server-model.fabro").to_str().unwrap(),
|
||||
])
|
||||
.output()
|
||||
.expect("command should execute");
|
||||
|
||||
assert!(
|
||||
output.status.success(),
|
||||
"local validation should not reject a server-owned provider\nstdout:\n{}\nstderr:\n{}",
|
||||
String::from_utf8_lossy(&output.stdout),
|
||||
String::from_utf8_lossy(&output.stderr)
|
||||
);
|
||||
mock.assert();
|
||||
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn create_uses_configured_server_target_without_server_flag() {
|
||||
let context = test_context!();
|
||||
|
|
|
|||
|
|
@ -95,7 +95,9 @@ fn graph_allow_invalid_renders_after_diagnostics() {
|
|||
----- stdout -----
|
||||
----- stderr -----
|
||||
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
|
||||
fix: Add a node with shape=Mdiamond or id 'start'
|
||||
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
|
||||
fix: Remove outgoing edges from the exit node
|
||||
");
|
||||
|
||||
let svg = read_text(&output_path);
|
||||
|
|
@ -119,7 +121,9 @@ fn graph_invalid_workflow_fails_after_diagnostics() {
|
|||
----- stdout -----
|
||||
----- stderr -----
|
||||
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
|
||||
fix: Add a node with shape=Mdiamond or id 'start'
|
||||
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
|
||||
fix: Remove outgoing edges from the exit node
|
||||
× Validation failed
|
||||
");
|
||||
}
|
||||
|
|
|
|||
|
|
@ -52,7 +52,9 @@ fn preflight_invalid_workflow_fails_with_validation_output() {
|
|||
Workflow: Invalid (2 nodes, 1 edges)
|
||||
Graph: [FIXTURES]/invalid.fabro
|
||||
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
|
||||
fix: Add a node with shape=Mdiamond or id 'start'
|
||||
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
|
||||
fix: Remove outgoing edges from the exit node
|
||||
× Validation failed
|
||||
");
|
||||
}
|
||||
|
|
@ -74,7 +76,9 @@ fn preflight_rejects_unbound_template_inputs() {
|
|||
Goal: Demo
|
||||
|
||||
error: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
|
||||
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
|
||||
error: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
|
||||
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
|
||||
× Validation failed
|
||||
");
|
||||
}
|
||||
|
|
|
|||
|
|
@ -69,6 +69,24 @@ fn simple_does_not_connect_to_configured_server() {
|
|||
);
|
||||
}
|
||||
|
||||
/// Offline validation has no model catalog, so a model or provider the server
|
||||
/// owns must pass rather than be reported as unknown.
|
||||
#[test]
|
||||
fn server_owned_provider_is_not_rejected_by_offline_validation() {
|
||||
let context = test_context!();
|
||||
let mut cmd = context.validate();
|
||||
cmd.arg(fixture("server-model.fabro"));
|
||||
fabro_snapshot!(context.filters(), cmd, @"
|
||||
success: true
|
||||
exit_code: 0
|
||||
----- stdout -----
|
||||
----- stderr -----
|
||||
Workflow: ServerModel (3 nodes, 2 edges)
|
||||
Graph: [FIXTURES]/server-model.fabro
|
||||
Validation: OK
|
||||
");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn branching() {
|
||||
let context = test_context!();
|
||||
|
|
@ -82,6 +100,7 @@ fn branching() {
|
|||
Workflow: Branch (6 nodes, 6 edges)
|
||||
Graph: [FIXTURES]/branching.fabro
|
||||
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
|
||||
fix: Add retry_target or fallback_retry_target attribute
|
||||
Validation: OK
|
||||
");
|
||||
}
|
||||
|
|
@ -163,7 +182,9 @@ fn bare_fabro_with_unbound_inputs_validates_structurally_with_warning() {
|
|||
Workflow: TemplatedUnbound (3 nodes, 2 edges)
|
||||
Graph: [FIXTURES]/templated_unbound.fabro
|
||||
warning: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
|
||||
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
|
||||
warning: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
|
||||
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
|
||||
Validation: OK
|
||||
");
|
||||
}
|
||||
|
|
@ -186,6 +207,7 @@ fn bare_fabro_with_unbound_inputs_in_imported_prompt_validates_structurally_with
|
|||
Workflow: TemplatedUnboundImported (3 nodes, 2 edges)
|
||||
Graph: [FIXTURES]/templated_unbound_imported/workflow.fabro
|
||||
warning: [FIXTURES]/templated_unbound_imported/work.md:1:12: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
|
||||
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
|
||||
Validation: OK
|
||||
");
|
||||
}
|
||||
|
|
@ -207,6 +229,7 @@ fn bare_fabro_with_unbound_inputs_in_template_partial_validates_structurally_wit
|
|||
Workflow: TemplatedUnboundPartial (3 nodes, 2 edges)
|
||||
Graph: [FIXTURES]/templated_unbound_partial/workflow.fabro
|
||||
warning: [FIXTURES]/templated_unbound_partial/test-include.partial.md:1:4: undefined template variable `inputs.hello` in node `test_imported_include` attribute `prompt` [node: test_imported_include] (template_undefined_variable)
|
||||
fix: bind `inputs.hello` via `[run.inputs]` in workflow.toml, or pass `--input inputs.hello=<value>`
|
||||
Validation: OK
|
||||
");
|
||||
}
|
||||
|
|
@ -278,6 +301,26 @@ fn validate_reports_missing_template_dependency() {
|
|||
");
|
||||
}
|
||||
|
||||
/// A node named only by an edge is almost always a typo, so validation must
|
||||
/// fail instead of quietly running it as a default agent stage.
|
||||
#[test]
|
||||
fn edge_only_node() {
|
||||
let context = test_context!();
|
||||
let mut cmd = context.validate();
|
||||
cmd.arg(fixture("edge_only_node.fabro"));
|
||||
fabro_snapshot!(context.filters(), cmd, @"
|
||||
success: false
|
||||
exit_code: 1
|
||||
----- stdout -----
|
||||
----- stderr -----
|
||||
Workflow: EdgeOnlyNode (2 nodes, 2 edges)
|
||||
Graph: [FIXTURES]/edge_only_node.fabro
|
||||
error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists)
|
||||
fix: Declare node 'misspelled_node' or correct the edge endpoint
|
||||
× Validation failed
|
||||
");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid() {
|
||||
let context = test_context!();
|
||||
|
|
@ -291,7 +334,9 @@ fn invalid() {
|
|||
Workflow: Invalid (2 nodes, 1 edges)
|
||||
Graph: [FIXTURES]/invalid.fabro
|
||||
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
|
||||
fix: Add a node with shape=Mdiamond or id 'start'
|
||||
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
|
||||
fix: Remove outgoing edges from the exit node
|
||||
× Validation failed
|
||||
");
|
||||
}
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ fn dry_run_branching() {
|
|||
Goal: Implement and validate a feature
|
||||
|
||||
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
|
||||
fix: Add retry_target or fallback_retry_target attribute
|
||||
Run: [ULID]
|
||||
Web UI: http://localhost:3000/runs/[ULID]
|
||||
Sandbox: local (ready in [TIME])
|
||||
|
|
|
|||
|
|
@ -1,11 +1,8 @@
|
|||
use std::path::Path;
|
||||
use std::sync::Arc;
|
||||
|
||||
use fabro_api::types;
|
||||
use fabro_config::load_llm_catalog_settings;
|
||||
use fabro_model::Catalog;
|
||||
use fabro_server::run_tool_manifest;
|
||||
use fabro_tool::{RunManifestBuilder, ToolError, ToolResult, ValidatedCreateRunSpec};
|
||||
use fabro_tool::{RunManifestBuilder, ToolResult, ValidatedCreateRunSpec};
|
||||
|
||||
#[derive(Default)]
|
||||
pub(crate) struct McpRunManifestBuilder;
|
||||
|
|
@ -17,20 +14,6 @@ impl RunManifestBuilder for McpRunManifestBuilder {
|
|||
cwd: &Path,
|
||||
user_settings_path: &Path,
|
||||
) -> ToolResult<types::RunManifest> {
|
||||
build_mcp_run_manifest(spec, cwd, user_settings_path)
|
||||
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
|
||||
}
|
||||
}
|
||||
|
||||
fn build_mcp_run_manifest(
|
||||
spec: &ValidatedCreateRunSpec,
|
||||
cwd: &Path,
|
||||
user_settings_path: &Path,
|
||||
) -> ToolResult<types::RunManifest> {
|
||||
let llm_catalog_settings = load_llm_catalog_settings(Some(user_settings_path))
|
||||
.map_err(|err| ToolError::message(err.to_string()))?;
|
||||
let catalog = Arc::new(
|
||||
Catalog::from_builtin_with_overrides(&llm_catalog_settings)
|
||||
.map_err(|err| ToolError::message(err.to_string()))?,
|
||||
);
|
||||
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, catalog)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1134,6 +1134,7 @@ mod runs {
|
|||
id: stage_id.clone(),
|
||||
name: name.to_owned(),
|
||||
handler,
|
||||
billing: BilledTokenCounts::default(),
|
||||
status,
|
||||
wall_time_ms,
|
||||
node_id: stage_id.node_id().to_owned(),
|
||||
|
|
|
|||
|
|
@ -1,41 +1,30 @@
|
|||
use std::collections::HashMap;
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::Result;
|
||||
use fabro_api::types;
|
||||
use fabro_config::{EnvironmentLayer, MergeMap, RunLayer};
|
||||
use fabro_model::Catalog;
|
||||
use fabro_config::RunLayer;
|
||||
use fabro_workflow::pipeline::TEMPLATE_UNDEFINED_VARIABLE_RULE;
|
||||
|
||||
use crate::run_manifest;
|
||||
|
||||
/// Validate a manifest without a model catalog.
|
||||
///
|
||||
/// Every caller is a client — the CLI, an MCP server, a run worker — and a
|
||||
/// client's catalog is its own, not the server's. Judging model and provider
|
||||
/// availability here would reject workflows the server can run, so that is
|
||||
/// left to the server on create.
|
||||
pub fn validate_manifest(
|
||||
manifest_run_defaults: &RunLayer,
|
||||
manifest: &types::RunManifest,
|
||||
catalog: Arc<Catalog>,
|
||||
) -> Result<types::ValidateResponse> {
|
||||
validate_manifest_with_environment_defaults(
|
||||
manifest_run_defaults,
|
||||
&fabro_environment::seeded_catalog_layer(),
|
||||
manifest,
|
||||
catalog,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn validate_manifest_with_environment_defaults(
|
||||
manifest_run_defaults: &RunLayer,
|
||||
manifest_environment_defaults: &MergeMap<EnvironmentLayer>,
|
||||
manifest: &types::RunManifest,
|
||||
catalog: Arc<Catalog>,
|
||||
) -> Result<types::ValidateResponse> {
|
||||
let prepared = run_manifest::prepare_manifest_with_environment_defaults(
|
||||
manifest_run_defaults,
|
||||
manifest_environment_defaults,
|
||||
&fabro_environment::seeded_catalog_layer(),
|
||||
&HashMap::new(),
|
||||
manifest,
|
||||
)?;
|
||||
let validated =
|
||||
run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?;
|
||||
let validated = run_manifest::validate_prepared_manifest_structural(&prepared)
|
||||
.map_err(anyhow::Error::new)?;
|
||||
Ok(run_manifest::validate_response(&prepared, &validated))
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -35,7 +35,8 @@ use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSecti
|
|||
use fabro_validate::Severity;
|
||||
use fabro_workflow::Error as WorkflowError;
|
||||
use fabro_workflow::operations::{
|
||||
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_ready_providers,
|
||||
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_catalog,
|
||||
validate_with_ready_providers,
|
||||
};
|
||||
use fabro_workflow::pipeline::Validated;
|
||||
use fabro_workflow::run_materialization::materialize_run_with_ready_providers;
|
||||
|
|
@ -192,12 +193,18 @@ pub(crate) fn validate_prepared_manifest(
|
|||
validate_prepared_manifest_with_vars(prepared, catalog, HashMap::new())
|
||||
}
|
||||
|
||||
pub(crate) fn validate_prepared_manifest_structural(
|
||||
prepared: &PreparedManifest,
|
||||
) -> Result<Validated, WorkflowError> {
|
||||
validate(manifest_validate_input(prepared, HashMap::new()))
|
||||
}
|
||||
|
||||
pub(crate) fn validate_prepared_manifest_with_vars(
|
||||
prepared: &PreparedManifest,
|
||||
catalog: Arc<Catalog>,
|
||||
vars: HashMap<String, String>,
|
||||
) -> Result<Validated, WorkflowError> {
|
||||
validate(manifest_validate_input(prepared, catalog, vars))
|
||||
validate_with_catalog(manifest_validate_input(prepared, vars), catalog)
|
||||
}
|
||||
|
||||
pub(crate) fn validate_prepared_manifest_for_preflight(
|
||||
|
|
@ -207,14 +214,14 @@ pub(crate) fn validate_prepared_manifest_for_preflight(
|
|||
ready_providers: &[ProviderId],
|
||||
) -> Result<Validated, WorkflowError> {
|
||||
validate_with_ready_providers(
|
||||
manifest_validate_input(prepared, catalog, vars),
|
||||
manifest_validate_input(prepared, vars),
|
||||
catalog,
|
||||
ready_providers,
|
||||
)
|
||||
}
|
||||
|
||||
fn manifest_validate_input(
|
||||
prepared: &PreparedManifest,
|
||||
catalog: Arc<Catalog>,
|
||||
vars: HashMap<String, String>,
|
||||
) -> ValidateInput {
|
||||
ValidateInput {
|
||||
|
|
@ -223,7 +230,6 @@ fn manifest_validate_input(
|
|||
vars,
|
||||
cwd: prepared.cwd.clone(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1,20 +1,22 @@
|
|||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Arc;
|
||||
|
||||
use fabro_api::types;
|
||||
use fabro_config::{CliLayer, RunGoalLayer, RunLayer};
|
||||
use fabro_manifest::{ManifestBuildInput, RunOverrideInput};
|
||||
use fabro_model::Catalog;
|
||||
use fabro_tool::{ToolError, ToolResult, ValidatedCreateRunSpec};
|
||||
use fabro_types::settings::interp::InterpString;
|
||||
|
||||
use crate::manifest_validation;
|
||||
|
||||
/// Build and validate a run manifest for the `fabro_run_create` tool.
|
||||
///
|
||||
/// Validation is structural. The caller is a client — an MCP server or a run
|
||||
/// worker — whose catalog is its own, not the server's, so judging model and
|
||||
/// provider availability here would reject workflows the server can run.
|
||||
pub fn build_run_tool_manifest(
|
||||
spec: &ValidatedCreateRunSpec,
|
||||
cwd: &Path,
|
||||
user_settings_path: &Path,
|
||||
catalog: Arc<Catalog>,
|
||||
) -> ToolResult<types::RunManifest> {
|
||||
let built = fabro_manifest::build_run_manifest(ManifestBuildInput {
|
||||
workflow: PathBuf::from(&spec.workflow),
|
||||
|
|
@ -30,7 +32,7 @@ pub fn build_run_tool_manifest(
|
|||
.map_err(|err| ToolError::from_anyhow(&err))?;
|
||||
|
||||
let mut validation =
|
||||
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest, catalog)
|
||||
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)
|
||||
.map_err(|err| ToolError::from_anyhow(&err))?;
|
||||
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
|
||||
if !validation.ok {
|
||||
|
|
@ -103,6 +105,63 @@ mod tests {
|
|||
|
||||
use super::*;
|
||||
|
||||
fn create_run_spec(workflow: &str) -> ValidatedCreateRunSpec {
|
||||
ValidatedCreateRunSpec::try_from(CreateRunSpec {
|
||||
workflow: workflow.to_string(),
|
||||
run_id: None,
|
||||
parent_id: None,
|
||||
cwd: None,
|
||||
goal: None,
|
||||
goal_file: None,
|
||||
inputs: HashMap::new(),
|
||||
labels: HashMap::new(),
|
||||
model: None,
|
||||
provider: None,
|
||||
environment: None,
|
||||
dry_run: None,
|
||||
auto_approve: None,
|
||||
preserve_sandbox: None,
|
||||
start: None,
|
||||
})
|
||||
.expect("create spec should validate")
|
||||
}
|
||||
|
||||
/// The tool runs on a client, whose catalog is not the server's, so a
|
||||
/// server-owned model must reach the server rather than fail here.
|
||||
#[expect(
|
||||
clippy::disallowed_methods,
|
||||
reason = "sync test writes one workflow fixture before building the manifest"
|
||||
)]
|
||||
#[test]
|
||||
fn server_owned_provider_is_not_rejected_by_tool_manifest_validation() {
|
||||
let dir = tempfile::tempdir().expect("temp dir should be created");
|
||||
let workflow = dir.path().join("server-model.fabro");
|
||||
std::fs::write(
|
||||
&workflow,
|
||||
r#"digraph ServerModel {
|
||||
graph [goal="Use a server-owned model"]
|
||||
start [shape=Mdiamond]
|
||||
work [prompt="Do work", model="private-model", provider="server-only"]
|
||||
exit [shape=Msquare]
|
||||
start -> work -> exit
|
||||
}"#,
|
||||
)
|
||||
.expect("workflow fixture should be written");
|
||||
|
||||
let manifest = build_run_tool_manifest(
|
||||
&create_run_spec(&workflow.to_string_lossy()),
|
||||
dir.path(),
|
||||
&dir.path().join("settings.toml"),
|
||||
)
|
||||
.expect("tool validation should leave provider availability to the server");
|
||||
|
||||
let encoded = serde_json::to_string(&manifest).expect("manifest should serialize");
|
||||
assert!(
|
||||
encoded.contains("server-only"),
|
||||
"the authored provider should survive into the manifest: {encoded}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn manifest_args_preserve_input_provenance() {
|
||||
let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec {
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ use std::collections::HashMap;
|
|||
use std::sync::Arc;
|
||||
|
||||
use chrono::{DateTime, Utc};
|
||||
use fabro_model::Catalog;
|
||||
use fabro_types::{
|
||||
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
|
||||
};
|
||||
|
|
@ -22,6 +23,7 @@ fn run_stage_from_projection(
|
|||
stage_id: &StageId,
|
||||
stage: &StageProjection,
|
||||
graph: &Graph,
|
||||
catalog: &Catalog,
|
||||
now: DateTime<Utc>,
|
||||
) -> RunStage {
|
||||
let handler = stage.handler.unwrap_or_else(|| {
|
||||
|
|
@ -36,6 +38,7 @@ fn run_stage_from_projection(
|
|||
id: stage_id.clone(),
|
||||
name: stage_id.node_id().to_owned(),
|
||||
handler,
|
||||
billing: stage.billed_usage(Some(catalog)).into_owned(),
|
||||
status: stage.effective_state(),
|
||||
wall_time_ms: stage.live_wall_time_ms(now),
|
||||
node_id: stage_id.node_id().to_owned(),
|
||||
|
|
@ -67,9 +70,10 @@ async fn list_run_stages(
|
|||
|
||||
let now = Utc::now();
|
||||
let graph = projection.spec().graph();
|
||||
let catalog = state.catalog();
|
||||
let stages = projection
|
||||
.iter_stages()
|
||||
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, now))
|
||||
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, &catalog, now))
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()
|
||||
|
|
@ -175,7 +179,7 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
|
|||
index
|
||||
});
|
||||
let row = &mut rows[index];
|
||||
let stage_timing = billing_stage_timing(stage, now);
|
||||
let stage_timing = stage.live_timing(now);
|
||||
row.timing = row.timing.saturating_add(&stage_timing);
|
||||
|
||||
if stage_id.visit() >= row.latest_visit {
|
||||
|
|
@ -188,20 +192,6 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
|
|||
rows
|
||||
}
|
||||
|
||||
/// Per-visit timing for a stage. For terminal visits, the stored breakdown is
|
||||
/// used directly. For in-flight visits, fall back to the live wall-clock since
|
||||
/// `started_at` (no active breakdown yet — that is only finalized at terminal
|
||||
/// event time in v1).
|
||||
fn billing_stage_timing(stage: &StageProjection, now: DateTime<Utc>) -> StageTiming {
|
||||
if let Some(timing) = stage.timing {
|
||||
return timing;
|
||||
}
|
||||
if let Some(live_wall) = stage.live_wall_time_ms(now) {
|
||||
return StageTiming::wall_only(live_wall);
|
||||
}
|
||||
StageTiming::default()
|
||||
}
|
||||
|
||||
fn stage_has_billing_row(stage: &StageProjection) -> bool {
|
||||
stage.completion.is_some()
|
||||
|| stage.timing.is_some()
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ use axum_extra::extract::Query as ExtraQuery;
|
|||
use base64::Engine as _;
|
||||
use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
|
||||
use bytes::Bytes;
|
||||
use chrono::Utc;
|
||||
use chrono::{DateTime, Utc};
|
||||
use fabro_api::types::{
|
||||
BoardColumn, RunManifest, SubmitAnswerRequest, UpdateRunParentRequest, UpdateRunRequest,
|
||||
};
|
||||
|
|
@ -23,7 +23,7 @@ use fabro_store::{
|
|||
};
|
||||
use fabro_types::settings::ResolveError;
|
||||
use fabro_types::{
|
||||
AutomationRef, Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
|
||||
AutomationRef, Principal, Run, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
|
||||
RunStatusKind, StageContextWindow, StageContextWindowStaleness,
|
||||
StageContextWindowUnavailableReason, StageHandler, StageModelUsage, StageProjection,
|
||||
SystemActorKind, WorkflowSettings, parse_blob_ref,
|
||||
|
|
@ -298,7 +298,7 @@ async fn validate_parent_link(
|
|||
}
|
||||
|
||||
async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
|
||||
match state.stores.run_summaries.get(run_id, Utc::now()).await {
|
||||
match run_summary_at(state, run_id, Utc::now()).await {
|
||||
Ok(Some(summary)) => (
|
||||
StatusCode::OK,
|
||||
Json(state.decorate_run_summary(summary).await),
|
||||
|
|
@ -311,6 +311,28 @@ async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
|
|||
}
|
||||
}
|
||||
|
||||
/// Read the durable summary and overlay its timing from the live projection.
|
||||
///
|
||||
/// The SQLite read model stores active timing as of the most recent event.
|
||||
/// An open inference or tool bracket keeps accruing between events, so detail
|
||||
/// reads need the projection's current estimate while the run is non-terminal.
|
||||
async fn run_summary_at(
|
||||
state: &AppState,
|
||||
run_id: &RunId,
|
||||
now: DateTime<Utc>,
|
||||
) -> fabro_store::Result<Option<Run>> {
|
||||
let Some(mut summary) = state.stores.run_summaries.get(run_id, now).await? else {
|
||||
return Ok(None);
|
||||
};
|
||||
if summary.timestamps.completed_at.is_none() {
|
||||
let projection = state.stores.runs.get_cached_projection(run_id).await?;
|
||||
if let Some(timing) = projection.and_then(|projection| projection.live_run_timing(now)) {
|
||||
summary.timing = Some(timing);
|
||||
}
|
||||
}
|
||||
Ok(Some(summary))
|
||||
}
|
||||
|
||||
async fn list_runs(
|
||||
_auth: RequiredRunManagementActor,
|
||||
State(state): State<Arc<AppState>>,
|
||||
|
|
@ -934,7 +956,7 @@ async fn get_run_status(
|
|||
RequireRunManagementTarget(id, _actor): RequireRunManagementTarget,
|
||||
State(state): State<Arc<AppState>>,
|
||||
) -> Response {
|
||||
match state.stores.run_summaries.get(&id, Utc::now()).await {
|
||||
match run_summary_at(&state, &id, Utc::now()).await {
|
||||
Ok(Some(run)) => {
|
||||
(StatusCode::OK, Json(state.decorate_run_summary(run).await)).into_response()
|
||||
}
|
||||
|
|
|
|||
|
|
@ -5259,6 +5259,64 @@ fn test_billed_usage(
|
|||
.unwrap()
|
||||
}
|
||||
|
||||
async fn create_billed_retry_run(state: &Arc<AppState>, run_id: RunId) {
|
||||
create_durable_run_with_events(state, run_id, &[
|
||||
workflow_event::Event::RunSubmitted {
|
||||
definition_blob: None,
|
||||
},
|
||||
workflow_event::Event::RunStarting,
|
||||
workflow_event::Event::RunRunning,
|
||||
])
|
||||
.await;
|
||||
|
||||
append_scoped_stage_event(
|
||||
state,
|
||||
run_id,
|
||||
"verify",
|
||||
1,
|
||||
&workflow_event::Event::StageFailed {
|
||||
node_id: "verify".to_string(),
|
||||
name: "Verify".to_string(),
|
||||
index: 1,
|
||||
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
|
||||
will_retry: true,
|
||||
timing: fabro_types::StageTiming::wall_only(1200),
|
||||
billing: Some(test_billed_usage("gpt-old", 100, 10)),
|
||||
actor: None,
|
||||
},
|
||||
)
|
||||
.await;
|
||||
append_scoped_stage_event(
|
||||
state,
|
||||
run_id,
|
||||
"verify",
|
||||
2,
|
||||
&workflow_event::Event::StageCompleted {
|
||||
node_id: "verify".to_string(),
|
||||
name: "Verify".to_string(),
|
||||
index: 1,
|
||||
timing: fabro_types::StageTiming::wall_only(800),
|
||||
status: "succeeded".to_string(),
|
||||
preferred_label: None,
|
||||
suggested_next_ids: Vec::new(),
|
||||
billing: Some(test_billed_usage("gpt-new", 200, 20)),
|
||||
failure: None,
|
||||
notes: None,
|
||||
files_touched: Vec::new(),
|
||||
context_updates: None,
|
||||
jump_to_node: None,
|
||||
context_values: None,
|
||||
node_visits: None,
|
||||
loop_failure_signatures: None,
|
||||
restart_failure_signatures: None,
|
||||
response: None,
|
||||
attempt: 2,
|
||||
max_attempts: 2,
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_run_stages_distinguishes_visits() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
|
|
@ -5503,6 +5561,50 @@ async fn list_run_stages_exposes_execution_identity_for_resumed_stage() {
|
|||
assert_eq!(second["resumed_from_stage_id"], "work@1");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn run_billing_includes_live_stage_timing_in_rows_and_totals() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
let app = crate::test_support::build_test_router(Arc::clone(&state));
|
||||
let run_id = RunId::new();
|
||||
create_durable_run_with_events(&state, run_id, &[
|
||||
workflow_event::Event::RunSubmitted {
|
||||
definition_blob: None,
|
||||
},
|
||||
workflow_event::Event::RunStarting,
|
||||
workflow_event::Event::RunRunning,
|
||||
workflow_run_started_event(run_id),
|
||||
])
|
||||
.await;
|
||||
append_scoped_stage_event(
|
||||
&state,
|
||||
run_id,
|
||||
"work",
|
||||
1,
|
||||
&stage_started_event("work", "command"),
|
||||
)
|
||||
.await;
|
||||
|
||||
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("GET")
|
||||
.uri(api(&format!("/runs/{run_id}/billing")))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let body = response_json!(response, StatusCode::OK).await;
|
||||
let stages = body["stages"].as_array().unwrap();
|
||||
|
||||
assert_eq!(stages.len(), 1);
|
||||
let row_timing = &stages[0]["timing"];
|
||||
assert!(row_timing["active_time_ms"].as_u64().unwrap() > 0);
|
||||
assert_eq!(row_timing["tool_time_ms"], row_timing["active_time_ms"]);
|
||||
assert_eq!(&body["totals"]["timing"], row_timing);
|
||||
}
|
||||
|
||||
/// `checkpoint.completed_nodes` records every visit, so a looped node appears
|
||||
/// once per re-entry. Billing must dedup so a retried node renders as one row
|
||||
/// and `runtime_secs` is summed across all visits exactly once.
|
||||
|
|
@ -5654,65 +5756,9 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
|
|||
let state = test_app_state_with_isolated_storage();
|
||||
let app = crate::test_support::build_test_router(Arc::clone(&state));
|
||||
let run_id = RunId::new();
|
||||
let failed_usage = test_billed_usage("gpt-old", 100, 10);
|
||||
|
||||
create_billed_retry_run(&state, run_id).await;
|
||||
let success_usage = test_billed_usage("gpt-new", 200, 20);
|
||||
|
||||
create_durable_run_with_events(&state, run_id, &[
|
||||
workflow_event::Event::RunSubmitted {
|
||||
definition_blob: None,
|
||||
},
|
||||
workflow_event::Event::RunStarting,
|
||||
workflow_event::Event::RunRunning,
|
||||
])
|
||||
.await;
|
||||
|
||||
append_scoped_stage_event(
|
||||
&state,
|
||||
run_id,
|
||||
"verify",
|
||||
1,
|
||||
&workflow_event::Event::StageFailed {
|
||||
node_id: "verify".to_string(),
|
||||
name: "Verify".to_string(),
|
||||
index: 1,
|
||||
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
|
||||
will_retry: true,
|
||||
timing: fabro_types::StageTiming::wall_only(1200),
|
||||
billing: Some(failed_usage),
|
||||
actor: None,
|
||||
},
|
||||
)
|
||||
.await;
|
||||
append_scoped_stage_event(
|
||||
&state,
|
||||
run_id,
|
||||
"verify",
|
||||
2,
|
||||
&workflow_event::Event::StageCompleted {
|
||||
node_id: "verify".to_string(),
|
||||
name: "Verify".to_string(),
|
||||
index: 1,
|
||||
timing: fabro_types::StageTiming::wall_only(800),
|
||||
status: "succeeded".to_string(),
|
||||
preferred_label: None,
|
||||
suggested_next_ids: Vec::new(),
|
||||
billing: Some(success_usage.clone()),
|
||||
failure: None,
|
||||
notes: None,
|
||||
files_touched: Vec::new(),
|
||||
context_updates: None,
|
||||
jump_to_node: None,
|
||||
context_values: None,
|
||||
node_visits: None,
|
||||
loop_failure_signatures: None,
|
||||
restart_failure_signatures: None,
|
||||
response: None,
|
||||
attempt: 2,
|
||||
max_attempts: 2,
|
||||
},
|
||||
)
|
||||
.await;
|
||||
|
||||
let mut latest_outcome: Outcome<Option<fabro_model::BilledModelUsage>> = Outcome::success();
|
||||
latest_outcome.usage = Some(success_usage);
|
||||
latest_outcome.timing = Some(fabro_types::StageTiming::wall_only(800));
|
||||
|
|
@ -5746,7 +5792,6 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
|
|||
.unwrap();
|
||||
|
||||
let response = app
|
||||
.clone()
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("GET")
|
||||
|
|
@ -5791,6 +5836,76 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
|
|||
assert_eq!(new_model["billing"]["input_tokens"], 200);
|
||||
}
|
||||
|
||||
/// The stage popover reads `billing` off the stages list, so it must be scoped
|
||||
/// to one visit — unlike the Billing tab's rows, which sum every visit of a
|
||||
/// node.
|
||||
#[tokio::test]
|
||||
async fn list_run_stages_reports_billing_per_visit() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
let app = crate::test_support::build_test_router(Arc::clone(&state));
|
||||
let run_id = RunId::new();
|
||||
|
||||
create_billed_retry_run(&state, run_id).await;
|
||||
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("GET")
|
||||
.uri(api(&format!("/runs/{run_id}/stages")))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let body = response_json!(response, StatusCode::OK).await;
|
||||
|
||||
let first = stage_entry(&body, "verify@1");
|
||||
assert_eq!(first["billing"]["input_tokens"], 100);
|
||||
assert_eq!(first["billing"]["output_tokens"], 10);
|
||||
assert_eq!(first["billing"]["total_usd_micros"], 110);
|
||||
|
||||
let second = stage_entry(&body, "verify@2");
|
||||
assert_eq!(second["billing"]["input_tokens"], 200);
|
||||
assert_eq!(second["billing"]["output_tokens"], 20);
|
||||
assert_eq!(second["billing"]["total_usd_micros"], 220);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
let app = crate::test_support::build_test_router(Arc::clone(&state));
|
||||
let run_id = RunId::new();
|
||||
|
||||
create_durable_run_with_events(&state, run_id, &[
|
||||
workflow_event::Event::RunSubmitted {
|
||||
definition_blob: None,
|
||||
},
|
||||
workflow_event::Event::RunStarting,
|
||||
workflow_event::Event::RunRunning,
|
||||
])
|
||||
.await;
|
||||
let started = stage_started_event("script", "command");
|
||||
append_scoped_stage_event(&state, run_id, "script", 1, &started).await;
|
||||
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("GET")
|
||||
.uri(api(&format!("/runs/{run_id}/stages")))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let body = response_json!(response, StatusCode::OK).await;
|
||||
|
||||
let billing = &stage_entry(&body, "script@1")["billing"];
|
||||
assert_eq!(billing["input_tokens"], 0);
|
||||
assert_eq!(billing["output_tokens"], 0);
|
||||
// No model ran, so there is nothing to price — not a $0.00 cost.
|
||||
assert!(billing.get("total_usd_micros").is_none());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_run_stages_shows_retrying_after_failed_event() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
|
|
@ -7800,6 +7915,51 @@ async fn get_run_status_returns_status() {
|
|||
assert!(body["labels"].is_object());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_run_status_advances_live_active_timing_between_events() {
|
||||
let state = test_app_state_with_isolated_storage();
|
||||
let app = crate::test_support::build_test_router(Arc::clone(&state));
|
||||
let run_id = RunId::new();
|
||||
create_durable_run_with_events(&state, run_id, &[
|
||||
workflow_event::Event::RunSubmitted {
|
||||
definition_blob: None,
|
||||
},
|
||||
workflow_event::Event::RunStarting,
|
||||
workflow_event::Event::RunRunning,
|
||||
workflow_run_started_event(run_id),
|
||||
])
|
||||
.await;
|
||||
append_scoped_stage_event(
|
||||
&state,
|
||||
run_id,
|
||||
"work",
|
||||
1,
|
||||
&stage_started_event("work", "command"),
|
||||
)
|
||||
.await;
|
||||
|
||||
// The SQLite summary stores timing at the StageStarted event. A later
|
||||
// detail read must overlay the in-flight command's active time from the
|
||||
// projection even though no newer event has arrived.
|
||||
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("GET")
|
||||
.uri(api(&format!("/runs/{run_id}")))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
let body = response_json!(response, StatusCode::OK).await;
|
||||
let timing = &body["timing"];
|
||||
|
||||
assert!(timing["active_time_ms"].as_u64().unwrap() > 0);
|
||||
assert_eq!(timing["tool_time_ms"], timing["active_time_ms"]);
|
||||
assert!(timing["wall_time_ms"].as_u64().unwrap() >= timing["active_time_ms"].as_u64().unwrap());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_run_status_not_found() {
|
||||
let app = test_app_with();
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ use axum::middleware::Next;
|
|||
use axum::response::Response;
|
||||
use axum::{Router, middleware};
|
||||
use chrono::Duration as ChronoDuration;
|
||||
use fabro_config::user::default_storage_dir;
|
||||
use fabro_config::{RunLayer, ServerSettingsBuilder, Storage, envfile};
|
||||
use fabro_db::DbPool;
|
||||
use fabro_interview::Interviewer;
|
||||
|
|
@ -253,9 +254,10 @@ impl TestAppStateBuilder {
|
|||
self.try_build().expect("test app state should build")
|
||||
}
|
||||
|
||||
pub fn try_build(self) -> anyhow::Result<Arc<AppState>> {
|
||||
pub fn try_build(mut self) -> anyhow::Result<Arc<AppState>> {
|
||||
let (store, artifact_store) = self.store_bundle.unwrap_or_else(test_store_bundle);
|
||||
let vault_path = self.vault_path.unwrap_or_else(test_secret_store_path);
|
||||
self.server_settings = redirect_default_storage_root(self.server_settings, &vault_path);
|
||||
if !self.vault_entries.is_empty() {
|
||||
let mut vault = Vault::load(vault_path.clone()).expect("test vault should load");
|
||||
for (name, value) in &self.vault_entries {
|
||||
|
|
@ -672,6 +674,27 @@ pub fn test_secret_store_path() -> PathBuf {
|
|||
dir.join("secrets.json")
|
||||
}
|
||||
|
||||
/// Keeps tests off the developer's real `~/.fabro/storage`.
|
||||
///
|
||||
/// Settings built for tests usually omit `[server.storage] root`, which
|
||||
/// resolves to the production default. Handlers that walk that tree — `df`,
|
||||
/// `system/resources`, `prune` — then read whatever runs and scratch
|
||||
/// directories the machine happens to have, making tests slow and
|
||||
/// machine-dependent, and letting run-creating tests write there.
|
||||
///
|
||||
/// Only settings still carrying the production default are redirected; a test
|
||||
/// that chose its own root keeps it. The redirect goes through
|
||||
/// [`ServerSettings::with_storage_override`] so the derived local object-store
|
||||
/// roots move with it instead of pointing back at the real storage tree.
|
||||
fn redirect_default_storage_root(settings: ServerSettings, vault_path: &Path) -> ServerSettings {
|
||||
if Path::new(&settings.server.storage.root) != default_storage_dir() {
|
||||
return settings;
|
||||
}
|
||||
let root = vault_path.with_file_name("storage");
|
||||
std::fs::create_dir_all(&root).expect("test storage root should be creatable");
|
||||
settings.with_storage_override(&root)
|
||||
}
|
||||
|
||||
#[must_use]
|
||||
pub fn test_auth_mode() -> AuthMode {
|
||||
AuthMode::Enabled(ConfiguredAuth {
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
use std::collections::HashMap;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::error::Error;
|
||||
|
|
@ -57,29 +57,44 @@ fn derive_class_from_label(label: &str) -> String {
|
|||
.collect()
|
||||
}
|
||||
|
||||
fn collect_declared_node_ids(statements: &[Statement], node_ids: &mut HashSet<String>) {
|
||||
for statement in statements {
|
||||
match statement {
|
||||
Statement::Node(node) => {
|
||||
node_ids.insert(node.id.clone());
|
||||
}
|
||||
Statement::Subgraph(subgraph) => {
|
||||
collect_declared_node_ids(&subgraph.statements, node_ids);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct SemanticState {
|
||||
graph: Graph,
|
||||
node_defaults: HashMap<String, AttrValue>,
|
||||
edge_defaults: HashMap<String, AttrValue>,
|
||||
graph: Graph,
|
||||
declared_node_ids: HashSet<String>,
|
||||
node_defaults: HashMap<String, AttrValue>,
|
||||
edge_defaults: HashMap<String, AttrValue>,
|
||||
}
|
||||
|
||||
impl SemanticState {
|
||||
fn new(name: String) -> Self {
|
||||
fn new(name: String, declared_node_ids: HashSet<String>) -> Self {
|
||||
Self {
|
||||
graph: Graph::new(name),
|
||||
graph: Graph::new(name),
|
||||
declared_node_ids,
|
||||
node_defaults: HashMap::new(),
|
||||
edge_defaults: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
fn ensure_node(&mut self, id: &str) {
|
||||
if !self.graph.nodes.contains_key(id) {
|
||||
fn ensure_node(&mut self, id: &str) -> &mut Node {
|
||||
let node_defaults = &self.node_defaults;
|
||||
self.graph.nodes.entry(id.to_string()).or_insert_with(|| {
|
||||
let mut node = Node::new(id);
|
||||
for (k, v) in &self.node_defaults {
|
||||
node.attrs.insert(k.clone(), v.clone());
|
||||
}
|
||||
self.graph.nodes.insert(id.to_string(), node);
|
||||
}
|
||||
node.attrs.clone_from(node_defaults);
|
||||
node
|
||||
})
|
||||
}
|
||||
|
||||
fn add_class_to_node(node: &mut Node, cls: &str) {
|
||||
|
|
@ -90,12 +105,7 @@ impl SemanticState {
|
|||
}
|
||||
|
||||
fn process_node(&mut self, node_stmt: &NodeStmt, subgraph_class: Option<&str>) {
|
||||
self.ensure_node(&node_stmt.id);
|
||||
let node = self
|
||||
.graph
|
||||
.nodes
|
||||
.get_mut(&node_stmt.id)
|
||||
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
|
||||
let node = self.ensure_node(&node_stmt.id);
|
||||
if let Some(attrs) = &node_stmt.attrs {
|
||||
for (k, v) in attrs {
|
||||
node.attrs.insert(k.clone(), convert_value(v));
|
||||
|
|
@ -111,11 +121,6 @@ impl SemanticState {
|
|||
.and_then(AttrValue::as_str)
|
||||
.map(String::from);
|
||||
if let Some(class_str) = class_str {
|
||||
let node = self
|
||||
.graph
|
||||
.nodes
|
||||
.get_mut(&node_stmt.id)
|
||||
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
|
||||
for cls in class_str.split(',') {
|
||||
let cls = cls.trim().to_string();
|
||||
if !cls.is_empty() && !node.classes.contains(&cls) {
|
||||
|
|
@ -127,12 +132,11 @@ impl SemanticState {
|
|||
|
||||
fn process_edge(&mut self, edge_stmt: &EdgeStmt, subgraph_class: Option<&str>) {
|
||||
for id in &edge_stmt.nodes {
|
||||
self.ensure_node(id);
|
||||
if !self.declared_node_ids.contains(id) {
|
||||
continue;
|
||||
}
|
||||
let node = self.ensure_node(id);
|
||||
if let Some(cls) = subgraph_class {
|
||||
let node =
|
||||
self.graph.nodes.get_mut(id).expect(
|
||||
"node was just inserted by ensure_node, so get_mut cannot return None",
|
||||
);
|
||||
Self::add_class_to_node(node, cls);
|
||||
}
|
||||
}
|
||||
|
|
@ -251,7 +255,9 @@ impl SemanticState {
|
|||
///
|
||||
/// Returns an error if the AST cannot be converted to a valid graph.
|
||||
pub fn ast_to_graph(dot: &DotGraph) -> Result<Graph, Error> {
|
||||
let mut state = SemanticState::new(dot.name.clone());
|
||||
let mut declared_node_ids = HashSet::new();
|
||||
collect_declared_node_ids(&dot.statements, &mut declared_node_ids);
|
||||
let mut state = SemanticState::new(dot.name.clone(), declared_node_ids);
|
||||
let empty = HashMap::new();
|
||||
state.process_statements(&dot.statements, None, &empty, &empty);
|
||||
Ok(state.graph)
|
||||
|
|
@ -515,7 +521,7 @@ mod tests {
|
|||
}
|
||||
|
||||
#[test]
|
||||
fn ast_to_graph_implicit_nodes_from_edges() {
|
||||
fn ast_to_graph_keeps_undeclared_edge_endpoints_out_of_nodes() {
|
||||
let dot = DotGraph {
|
||||
name: "Implicit".into(),
|
||||
statements: vec![Statement::Edge(EdgeStmt {
|
||||
|
|
@ -524,8 +530,83 @@ mod tests {
|
|||
})],
|
||||
};
|
||||
|
||||
let graph = ast_to_graph(&dot).unwrap();
|
||||
assert!(graph.nodes.is_empty());
|
||||
assert_eq!(graph.edges, vec![Edge::new("a", "b")]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ast_to_graph_includes_only_declared_edge_endpoints() {
|
||||
let dot = DotGraph {
|
||||
name: "Declared".into(),
|
||||
statements: vec![
|
||||
Statement::Node(NodeStmt {
|
||||
id: "a".into(),
|
||||
attrs: None,
|
||||
}),
|
||||
Statement::Edge(EdgeStmt {
|
||||
nodes: vec!["a".into(), "b".into()],
|
||||
attrs: None,
|
||||
}),
|
||||
],
|
||||
};
|
||||
|
||||
let graph = ast_to_graph(&dot).unwrap();
|
||||
assert!(graph.nodes.contains_key("a"));
|
||||
assert!(!graph.nodes.contains_key("b"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ast_to_graph_declaration_after_edge_still_counts() {
|
||||
let dot = DotGraph {
|
||||
name: "DeclaredLater".into(),
|
||||
statements: vec![
|
||||
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("first".into()))]),
|
||||
Statement::Edge(EdgeStmt {
|
||||
nodes: vec!["a".into(), "b".into()],
|
||||
attrs: None,
|
||||
}),
|
||||
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("second".into()))]),
|
||||
Statement::Node(NodeStmt {
|
||||
id: "b".into(),
|
||||
attrs: Some(vec![("prompt".into(), AstValue::Str("Do it".into()))]),
|
||||
}),
|
||||
],
|
||||
};
|
||||
|
||||
let graph = ast_to_graph(&dot).unwrap();
|
||||
assert!(graph.nodes.contains_key("b"));
|
||||
assert!(!graph.nodes.contains_key("a"));
|
||||
assert_eq!(
|
||||
graph.nodes["b"]
|
||||
.attrs
|
||||
.get("model")
|
||||
.and_then(AttrValue::as_str),
|
||||
Some("first")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ast_to_graph_subgraph_declaration_counts() {
|
||||
let dot = DotGraph {
|
||||
name: "SubgraphDeclared".into(),
|
||||
statements: vec![
|
||||
Statement::Edge(EdgeStmt {
|
||||
nodes: vec!["start".into(), "plan".into()],
|
||||
attrs: None,
|
||||
}),
|
||||
Statement::Subgraph(SubgraphStmt {
|
||||
name: Some("cluster_loop".into()),
|
||||
statements: vec![Statement::Node(NodeStmt {
|
||||
id: "plan".into(),
|
||||
attrs: None,
|
||||
})],
|
||||
}),
|
||||
],
|
||||
};
|
||||
|
||||
let graph = ast_to_graph(&dot).unwrap();
|
||||
assert!(graph.nodes.contains_key("plan"));
|
||||
assert!(!graph.nodes.contains_key("start"));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ use fabro_types::{
|
|||
SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection,
|
||||
StageModelUsage, StageOutcome, StageProjection, StageState, StartRecord, SubAgentProjection,
|
||||
SubAgentStatus, TodoListKind, TodoListProjection, TodoProjection, WorkflowRef, first_event_seq,
|
||||
timing,
|
||||
};
|
||||
use fabro_util::error::render_compact_with_causes;
|
||||
|
||||
|
|
@ -405,7 +406,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
};
|
||||
stage.response = response;
|
||||
stage.completion = Some(completion);
|
||||
stage.timing = Some(props.timing);
|
||||
stage.set_authoritative_timing(props.timing);
|
||||
if let Some(billing) = &props.billing {
|
||||
stage.usage.replace_with_billed_usage(billing);
|
||||
stage.model = Some(billing.model().clone());
|
||||
|
|
@ -428,7 +429,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
failure_reason,
|
||||
timestamp: ts,
|
||||
});
|
||||
stage.timing = Some(props.timing);
|
||||
stage.set_authoritative_timing(props.timing);
|
||||
if let Some(billing) = &props.billing {
|
||||
stage.usage.replace_with_billed_usage(billing);
|
||||
stage.model = Some(billing.model().clone());
|
||||
|
|
@ -449,7 +450,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
context_window.event_seq = Some(event.seq);
|
||||
stage.context_window = Some(context_window);
|
||||
}
|
||||
close_inference_bracket(self, stored, props.visit, event.seq);
|
||||
close_inference_bracket(self, stored, props.visit, event.seq, ts);
|
||||
}
|
||||
EventBody::AgentLlmStarted(props) => {
|
||||
open_inference_bracket(self, stored, props, event.seq, ts);
|
||||
|
|
@ -478,10 +479,10 @@ impl RunProjectionReducer for RunProjection {
|
|||
inference.first_output_kind = None;
|
||||
}
|
||||
EventBody::AgentError(props) => {
|
||||
close_inference_bracket(self, stored, props.visit, event.seq);
|
||||
close_inference_bracket(self, stored, props.visit, event.seq, ts);
|
||||
}
|
||||
EventBody::AgentSessionEnded(_) => {
|
||||
close_inference_brackets_for_session(self, stored);
|
||||
close_active_brackets_for_session(self, stored, ts);
|
||||
}
|
||||
EventBody::AgentSessionActivated(props) => {
|
||||
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
|
||||
|
|
@ -497,7 +498,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
return Ok(());
|
||||
};
|
||||
stage.agent_control = AgentControlState::WaitingForSteer;
|
||||
close_inference_bracket(self, stored, props.visit, event.seq);
|
||||
close_inference_bracket(self, stored, props.visit, event.seq, ts);
|
||||
}
|
||||
EventBody::AgentSteeringInjected(props) => {
|
||||
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
|
||||
|
|
@ -520,12 +521,17 @@ impl RunProjectionReducer for RunProjection {
|
|||
};
|
||||
stage.agent_tools.clone_from(&props.tools);
|
||||
}
|
||||
// `AgentAcpStarted` is the start-of-process signal for an external
|
||||
// ACP agent. `provider_used` is intentionally sourced from the
|
||||
// subsequent `AgentSessionActivated` event, which carries the
|
||||
// canonical provider/model. ACP runs without a steering hub never
|
||||
// emit activation and so legitimately leave `provider_used`
|
||||
// unset — matching legacy ACP behavior.
|
||||
EventBody::AgentAcpStarted(props) => {
|
||||
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
stage.open_acp_inference(ts);
|
||||
// `provider_used` is intentionally sourced from the subsequent
|
||||
// `AgentSessionActivated` event, which carries the canonical
|
||||
// provider/model. ACP runs without a steering hub never emit
|
||||
// activation and legitimately leave it unset.
|
||||
}
|
||||
EventBody::CommandStarted(props) => {
|
||||
let script_invocation = serde_json::to_value(props).map_err(|err| {
|
||||
Error::InvalidEvent(format!("invalid command.started payload: {err}"))
|
||||
|
|
@ -552,6 +558,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
|
||||
return Ok(());
|
||||
};
|
||||
stage.close_acp_inference(props.duration_ms);
|
||||
apply_agent_terminal(
|
||||
"agent.acp",
|
||||
stage,
|
||||
|
|
@ -564,6 +571,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
|
||||
return Ok(());
|
||||
};
|
||||
stage.close_acp_inference(props.duration_ms);
|
||||
apply_agent_terminal(
|
||||
"agent.acp",
|
||||
stage,
|
||||
|
|
@ -576,6 +584,7 @@ impl RunProjectionReducer for RunProjection {
|
|||
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
|
||||
return Ok(());
|
||||
};
|
||||
stage.close_acp_inference(props.duration_ms);
|
||||
apply_agent_terminal(
|
||||
"agent.acp",
|
||||
stage,
|
||||
|
|
@ -594,6 +603,11 @@ impl RunProjectionReducer for RunProjection {
|
|||
// Branches bypass the engine's StageStarted/StageCompleted
|
||||
// lifecycle. Seed started_at so the branch stage drives a live
|
||||
// wall-clock timer while it runs (the entry is created Running).
|
||||
let handler = stored
|
||||
.node_id
|
||||
.as_deref()
|
||||
.and_then(|node_id| self.spec().graph.nodes.get(node_id))
|
||||
.map(|node| StageHandler::from_handler_type(node.handler_type()));
|
||||
let is_new = stored
|
||||
.stage_id
|
||||
.as_ref()
|
||||
|
|
@ -604,6 +618,9 @@ impl RunProjectionReducer for RunProjection {
|
|||
if stage.started_at.is_none() {
|
||||
stage.started_at = Some(ts);
|
||||
}
|
||||
if stage.handler.is_none() {
|
||||
stage.handler = handler;
|
||||
}
|
||||
if is_new {
|
||||
stage.graph_visit = props.graph_visit;
|
||||
stage
|
||||
|
|
@ -746,6 +763,11 @@ impl RunProjectionReducer for RunProjection {
|
|||
});
|
||||
}
|
||||
EventBody::AgentToolStarted(props) => {
|
||||
let root_session_id = if stored.parent_session_id.is_none() {
|
||||
stored.session_id.clone()
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
|
||||
else {
|
||||
return Ok(());
|
||||
|
|
@ -766,6 +788,25 @@ impl RunProjectionReducer for RunProjection {
|
|||
projection.invoked = true;
|
||||
}
|
||||
}
|
||||
// A subagent's tools run inside the root session's tool call,
|
||||
// so the root batch already covers them. Timing them again
|
||||
// would double-count that span.
|
||||
if let Some(session_id) = root_session_id {
|
||||
stage.open_tool_call(session_id, props.tool_call_id.clone(), ts);
|
||||
}
|
||||
}
|
||||
EventBody::AgentToolCompleted(props) => {
|
||||
if stored.parent_session_id.is_some() {
|
||||
return Ok(());
|
||||
}
|
||||
let Some(session_id) = stored.session_id.as_deref() else {
|
||||
return Ok(());
|
||||
};
|
||||
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
stage.close_tool_call(session_id, &props.tool_call_id, ts);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
|
|
@ -1031,19 +1072,19 @@ fn open_inference_bracket(
|
|||
});
|
||||
}
|
||||
|
||||
/// Resolve the stage's `inference` slot when it holds a bracket this event is
|
||||
/// allowed to mutate.
|
||||
/// Resolve the stage owning a bracket this event is allowed to mutate,
|
||||
/// borrowing the whole projection so the caller can also fold elapsed time
|
||||
/// into the stage's live accumulators.
|
||||
///
|
||||
/// `None` when the event came from a child session, when the stage has no
|
||||
/// open bracket, or when the bracket belongs to a different session — the
|
||||
/// last case matters after failover, which discards the session and builds a
|
||||
/// new one within a single visit.
|
||||
fn matching_inference_slot<'a>(
|
||||
/// Returns `None` for child-session events, for a stage with no open bracket,
|
||||
/// and for a bracket belonging to a different session (which is what
|
||||
/// post-failover events look like).
|
||||
fn matching_inference_stage<'a>(
|
||||
state: &'a mut RunProjection,
|
||||
stored: &RunEvent,
|
||||
visit: u32,
|
||||
seq: u32,
|
||||
) -> Option<&'a mut Option<StageInferenceProjection>> {
|
||||
) -> Option<&'a mut StageProjection> {
|
||||
if stored.parent_session_id.is_some() {
|
||||
return None;
|
||||
}
|
||||
|
|
@ -1053,7 +1094,7 @@ fn matching_inference_slot<'a>(
|
|||
.inference
|
||||
.as_ref()
|
||||
.is_some_and(|inference| inference.session_id == session_id);
|
||||
opened_here.then_some(&mut stage.inference)
|
||||
opened_here.then_some(stage)
|
||||
}
|
||||
|
||||
/// Resolve the open inference bracket this event is allowed to mutate.
|
||||
|
|
@ -1063,17 +1104,39 @@ fn matching_inference_bracket<'a>(
|
|||
visit: u32,
|
||||
seq: u32,
|
||||
) -> Option<&'a mut StageInferenceProjection> {
|
||||
matching_inference_slot(state, stored, visit, seq)?.as_mut()
|
||||
matching_inference_stage(state, stored, visit, seq)?
|
||||
.inference
|
||||
.as_mut()
|
||||
}
|
||||
|
||||
/// Close the bracket on a stage-addressed terminal event.
|
||||
fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: u32, seq: u32) {
|
||||
if let Some(slot) = matching_inference_slot(state, stored, visit, seq) {
|
||||
*slot = None;
|
||||
}
|
||||
/// Close the bracket on a stage-addressed terminal event, folding its elapsed
|
||||
/// time into the stage's live inference accumulator.
|
||||
fn close_inference_bracket(
|
||||
state: &mut RunProjection,
|
||||
stored: &RunEvent,
|
||||
visit: u32,
|
||||
seq: u32,
|
||||
ts: DateTime<Utc>,
|
||||
) {
|
||||
let Some(stage) = matching_inference_stage(state, stored, visit, seq) else {
|
||||
return;
|
||||
};
|
||||
close_bracket_on_stage(stage, ts);
|
||||
}
|
||||
|
||||
/// Close every bracket opened by the session that just ended.
|
||||
/// Take the open bracket and add its span to `live_inference_ms`.
|
||||
///
|
||||
/// Retries inside the bracket are deliberately included: the in-process
|
||||
/// stopwatch counts a retried attempt's elapsed time as inference, and
|
||||
/// `agent.llm.retry` keeps the bracket open rather than reopening it.
|
||||
fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime<Utc>) {
|
||||
let Some(inference) = stage.inference.take() else {
|
||||
return;
|
||||
};
|
||||
stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, ts));
|
||||
}
|
||||
|
||||
/// Close every active bracket opened by the session that just ended.
|
||||
///
|
||||
/// `agent.session.ended` is the only ordering-safe backstop for terminal
|
||||
/// cancel and wall-clock timeout, which tear the session down through
|
||||
|
|
@ -1088,7 +1151,11 @@ fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit:
|
|||
/// opened. Implemented as a normal stage lookup it would find no target and
|
||||
/// silently no-op, leaving the bracket open forever on exactly the path it
|
||||
/// exists to cover.
|
||||
fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunEvent) {
|
||||
fn close_active_brackets_for_session(
|
||||
state: &mut RunProjection,
|
||||
stored: &RunEvent,
|
||||
ts: DateTime<Utc>,
|
||||
) {
|
||||
if stored.parent_session_id.is_some() {
|
||||
return;
|
||||
}
|
||||
|
|
@ -1101,8 +1168,9 @@ fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunE
|
|||
.as_ref()
|
||||
.is_some_and(|inference| inference.session_id == session_id);
|
||||
if opened_here {
|
||||
stage.inference = None;
|
||||
close_bracket_on_stage(stage, ts);
|
||||
}
|
||||
stage.close_tool_batch_for_session(session_id, ts);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -1412,23 +1480,26 @@ fn finalize_unfinished_stages_after_run_failed(
|
|||
StageState::Failed
|
||||
};
|
||||
|
||||
for (_, stage) in state.iter_stages_mut() {
|
||||
for (_, stage) in state.iter_stages_unordered_mut() {
|
||||
if stage.state.is_terminal() {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Close any brackets still open so their spans are not dropped on the
|
||||
// floor when the live estimate is frozen into `timing` below.
|
||||
close_bracket_on_stage(stage, timestamp);
|
||||
stage.close_open_acp_inference(timestamp);
|
||||
stage.close_open_tool_batch(timestamp);
|
||||
|
||||
// Freeze the live estimate before flipping to a terminal state:
|
||||
// `live_timing` reads `effective_state` and would return wall-only
|
||||
// once the stage no longer looks in-flight.
|
||||
let frozen = stage.live_timing(timestamp);
|
||||
stage.state = terminal_state;
|
||||
if stage.timing.is_none() {
|
||||
if let Some(started_at) = stage.started_at {
|
||||
let wall_time_ms = u64::try_from(
|
||||
timestamp
|
||||
.signed_duration_since(started_at)
|
||||
.num_milliseconds()
|
||||
.max(0),
|
||||
)
|
||||
.expect("non-negative milliseconds fit in u64");
|
||||
stage.timing = Some(fabro_types::StageTiming::wall_only(wall_time_ms));
|
||||
}
|
||||
if stage.timing.is_none() && stage.started_at.is_some() {
|
||||
stage.set_authoritative_timing(frozen);
|
||||
} else {
|
||||
stage.clear_live_timing();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -1545,21 +1616,519 @@ mod tests {
|
|||
};
|
||||
use fabro_types::settings::run::{DockerfileSource, EnvironmentProvider};
|
||||
use fabro_types::{
|
||||
AgentBackend, AgentControlState, AutomationRef, BilledModelUsage, BilledTokenCounts,
|
||||
BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody,
|
||||
FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Outcome,
|
||||
PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
|
||||
AgentBackend, AgentControlState, AttrValue, AutomationRef, BilledModelUsage,
|
||||
BilledTokenCounts, BlockedReason, Checkpoint, CheckpointRecord, CommandTermination,
|
||||
EventBody, FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Node,
|
||||
Outcome, PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
|
||||
RunApprovalState, RunBlobId, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec,
|
||||
RunStatus, Speed, StageContextWindowBreakdownItem, StageContextWindowCategory,
|
||||
StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
|
||||
StageContextWindowWarning, StageModelUsage, StageOutcome, StageState, SubAgentStatus,
|
||||
SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support,
|
||||
StageContextWindowWarning, StageHandler, StageModelUsage, StageOutcome, StageState,
|
||||
StageTiming, SubAgentStatus, SuccessReason, WorkflowSettings, first_event_seq, fixtures,
|
||||
test_support,
|
||||
};
|
||||
use serde_json::json;
|
||||
|
||||
use super::{RunProjection, RunProjectionReducer, build_summary};
|
||||
use crate::{Error, EventEnvelope, StageId};
|
||||
|
||||
/// Live accumulation of inference and tool time while a stage is in
|
||||
/// flight. The finalized breakdown still arrives with the terminal event
|
||||
/// and replaces these; these exist so a long-running stage is not reported
|
||||
/// as doing no work.
|
||||
mod live_active_accumulation {
|
||||
use fabro_types::run_event::{
|
||||
AgentLlmFirstOutputProps, AgentLlmRetryProps, AgentLlmStartedProps,
|
||||
AgentToolCompletedProps, AgentToolStartedProps,
|
||||
};
|
||||
use fabro_types::{
|
||||
LlmOutputKind, LlmRetryPhase, ModelRef, Speed, StageOutcome, StageProjection,
|
||||
};
|
||||
|
||||
use super::*;
|
||||
|
||||
fn stage_id() -> StageId {
|
||||
StageId::new("plan", 1)
|
||||
}
|
||||
|
||||
fn session_event(seq: u32, ts: &str, session_id: &str, body: EventBody) -> EventEnvelope {
|
||||
let mut event = test_stage_event_at(seq, ts, body, stage_id());
|
||||
event.event.session_id = Some(session_id.to_string());
|
||||
event
|
||||
}
|
||||
|
||||
fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
|
||||
session_event(seq, ts, "session-1", body)
|
||||
}
|
||||
|
||||
/// An event from a sub-agent session nested under the root session.
|
||||
fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
|
||||
let mut event = agent_event(seq, ts, body);
|
||||
event.event.session_id = Some("session-child".to_string());
|
||||
event.event.parent_session_id = Some("session-1".to_string());
|
||||
event
|
||||
}
|
||||
|
||||
fn llm_started() -> EventBody {
|
||||
EventBody::AgentLlmStarted(AgentLlmStartedProps {
|
||||
requested_model: ModelRef {
|
||||
provider: "anthropic".parse().unwrap(),
|
||||
model_id: "claude-fable-5".into(),
|
||||
speed: Some(Speed::Fast),
|
||||
},
|
||||
visit: 1,
|
||||
})
|
||||
}
|
||||
|
||||
fn tool_started(tool_call_id: &str) -> EventBody {
|
||||
EventBody::AgentToolStarted(AgentToolStartedProps {
|
||||
tool_name: "Bash".to_string(),
|
||||
tool_call_id: tool_call_id.to_string(),
|
||||
arguments: json!({}),
|
||||
visit: 1,
|
||||
tool_call: None,
|
||||
turn_id: None,
|
||||
parent_message_id: None,
|
||||
})
|
||||
}
|
||||
|
||||
fn tool_completed(tool_call_id: &str) -> EventBody {
|
||||
EventBody::AgentToolCompleted(AgentToolCompletedProps {
|
||||
tool_name: "Bash".to_string(),
|
||||
tool_call_id: tool_call_id.to_string(),
|
||||
output: json!("ok"),
|
||||
is_error: false,
|
||||
visit: 1,
|
||||
tool_result: None,
|
||||
turn_id: None,
|
||||
})
|
||||
}
|
||||
|
||||
fn agent_message() -> EventBody {
|
||||
EventBody::AgentMessage(live_agent_message_props(live_counts(10, 5)))
|
||||
}
|
||||
|
||||
fn started_state() -> RunProjection {
|
||||
let mut state = initialized_projection();
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
1,
|
||||
"2026-04-07T12:00:00Z",
|
||||
EventBody::StageStarted(started_props()),
|
||||
stage_id(),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
}
|
||||
|
||||
fn stage(state: &RunProjection) -> &StageProjection {
|
||||
state.stage(&stage_id()).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn closing_an_inference_bracket_accumulates_its_span() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
3,
|
||||
"2026-04-07T12:00:06Z",
|
||||
EventBody::AgentLlmFirstOutput(AgentLlmFirstOutputProps {
|
||||
kind: LlmOutputKind::Text,
|
||||
visit: 1,
|
||||
}),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(4, "2026-04-07T12:00:12Z", agent_message()))
|
||||
.unwrap();
|
||||
|
||||
// 12:00:05 -> 12:00:12; first_output is a marker, not the close.
|
||||
assert_eq!(stage(&state).live_inference_ms, 7_000);
|
||||
assert!(stage(&state).inference.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrent_tool_calls_count_once_not_per_call() {
|
||||
let mut state = started_state();
|
||||
for (seq, id) in [(2, "call-a"), (3, "call-b"), (4, "call-c")] {
|
||||
state
|
||||
.apply_event(&agent_event(seq, "2026-04-07T12:00:00Z", tool_started(id)))
|
||||
.unwrap();
|
||||
}
|
||||
// All three finish 10s later. Summing per-call spans would report
|
||||
// 30s; the batch actually occupied 10s of wall time.
|
||||
for (seq, id) in [(5, "call-a"), (6, "call-b"), (7, "call-c")] {
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
seq,
|
||||
"2026-04-07T12:00:10Z",
|
||||
tool_completed(id),
|
||||
))
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
assert_eq!(stage(&state).live_tool_ms, 10_000);
|
||||
assert!(stage(&state).tool_batch.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_batch_stays_open_until_its_last_call_reports() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
2,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
3,
|
||||
"2026-04-07T12:00:02Z",
|
||||
tool_started("call-b"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
4,
|
||||
"2026-04-07T12:00:05Z",
|
||||
tool_completed("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
stage(&state).live_tool_ms,
|
||||
0,
|
||||
"batch must not close while call-b is outstanding"
|
||||
);
|
||||
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
5,
|
||||
"2026-04-07T12:00:09Z",
|
||||
tool_completed("call-b"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
// Measured from the batch open, not from the last call's start.
|
||||
assert_eq!(stage(&state).live_tool_ms, 9_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn successive_batches_accumulate() {
|
||||
let mut state = started_state();
|
||||
for (seq, ts, body) in [
|
||||
(2, "2026-04-07T12:00:00Z", tool_started("call-a")),
|
||||
(3, "2026-04-07T12:00:04Z", tool_completed("call-a")),
|
||||
(4, "2026-04-07T12:00:10Z", tool_started("call-b")),
|
||||
(5, "2026-04-07T12:00:16Z", tool_completed("call-b")),
|
||||
] {
|
||||
state.apply_event(&agent_event(seq, ts, body)).unwrap();
|
||||
}
|
||||
|
||||
assert_eq!(stage(&state).live_tool_ms, 10_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_duplicate_completion_does_not_drain_the_batch_early() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
2,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
3,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("call-b"),
|
||||
))
|
||||
.unwrap();
|
||||
// call-a reports twice, as a replayed or duplicated log can.
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
4,
|
||||
"2026-04-07T12:00:03Z",
|
||||
tool_completed("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
5,
|
||||
"2026-04-07T12:00:04Z",
|
||||
tool_completed("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(stage(&state).live_tool_ms, 0);
|
||||
assert!(stage(&state).tool_batch.is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_foreign_session_completion_does_not_mutate_the_open_batch() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
2,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&session_event(
|
||||
3,
|
||||
"2026-04-07T12:00:05Z",
|
||||
"session-2",
|
||||
tool_completed("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let batch = stage(&state).tool_batch.as_ref().unwrap();
|
||||
assert_eq!(batch.session_id, "session-1");
|
||||
assert!(batch.open_call_ids.contains("call-a"));
|
||||
assert_eq!(stage(&state).live_tool_ms, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_replacement_session_starts_a_separate_tool_batch() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
2,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("old-call"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&session_event(
|
||||
3,
|
||||
"2026-04-07T12:00:05Z",
|
||||
"session-2",
|
||||
tool_started("new-call"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(stage(&state).live_tool_ms, 5_000);
|
||||
let batch = stage(&state).tool_batch.as_ref().unwrap();
|
||||
assert_eq!(batch.session_id, "session-2");
|
||||
assert_eq!(
|
||||
batch.open_call_ids,
|
||||
["new-call".to_string()].into_iter().collect()
|
||||
);
|
||||
|
||||
// A delayed completion from the old session cannot close the new
|
||||
// session's batch even when call ids happen to collide.
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
4,
|
||||
"2026-04-07T12:00:07Z",
|
||||
tool_completed("new-call"),
|
||||
))
|
||||
.unwrap();
|
||||
assert!(stage(&state).tool_batch.is_some());
|
||||
|
||||
state
|
||||
.apply_event(&session_event(
|
||||
5,
|
||||
"2026-04-07T12:00:09Z",
|
||||
"session-2",
|
||||
tool_completed("new-call"),
|
||||
))
|
||||
.unwrap();
|
||||
assert_eq!(stage(&state).live_tool_ms, 9_000);
|
||||
assert!(stage(&state).tool_batch.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subagent_tool_calls_do_not_double_count_against_the_root_batch() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
2,
|
||||
"2026-04-07T12:00:00Z",
|
||||
tool_started("root-call"),
|
||||
))
|
||||
.unwrap();
|
||||
// The sub-agent's own tools run inside the root call's span.
|
||||
state
|
||||
.apply_event(&child_event(
|
||||
3,
|
||||
"2026-04-07T12:00:01Z",
|
||||
tool_started("child-call"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&child_event(
|
||||
4,
|
||||
"2026-04-07T12:00:02Z",
|
||||
tool_completed("child-call"),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
5,
|
||||
"2026-04-07T12:00:08Z",
|
||||
tool_completed("root-call"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(stage(&state).live_tool_ms, 8_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn session_end_accumulates_every_open_active_bracket() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
3,
|
||||
"2026-04-07T12:00:07Z",
|
||||
tool_started("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let mut ended = test_stage_event_at(
|
||||
4,
|
||||
"2026-04-07T12:00:20Z",
|
||||
EventBody::AgentSessionEnded(AgentSessionEndedProps {}),
|
||||
stage_id(),
|
||||
);
|
||||
ended.event.session_id = Some("session-1".to_string());
|
||||
state.apply_event(&ended).unwrap();
|
||||
|
||||
assert_eq!(stage(&state).live_inference_ms, 15_000);
|
||||
assert_eq!(stage(&state).live_tool_ms, 13_000);
|
||||
assert!(stage(&state).inference.is_none());
|
||||
assert!(stage(&state).tool_batch.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_foreign_session_close_leaves_the_bracket_and_accumulator_alone() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
|
||||
.unwrap();
|
||||
|
||||
// Post-failover: a new session emits the message, so the old
|
||||
// bracket is not this event's to close or bill.
|
||||
let mut foreign =
|
||||
test_stage_event_at(3, "2026-04-07T12:00:20Z", agent_message(), stage_id());
|
||||
foreign.event.session_id = Some("session-2".to_string());
|
||||
state.apply_event(&foreign).unwrap();
|
||||
|
||||
assert_eq!(stage(&state).live_inference_ms, 0);
|
||||
assert!(stage(&state).inference.is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_retry_keeps_accumulating_within_one_bracket() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
3,
|
||||
"2026-04-07T12:00:04Z",
|
||||
EventBody::AgentLlmRetry(AgentLlmRetryProps {
|
||||
provider: "anthropic".to_string(),
|
||||
model: "claude-fable-5".to_string(),
|
||||
attempt: 0,
|
||||
delay_secs: 0.0,
|
||||
error: json!({ "kind": "stream" }),
|
||||
phase: Some(LlmRetryPhase::Consume),
|
||||
visit: 1,
|
||||
}),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(4, "2026-04-07T12:00:11Z", agent_message()))
|
||||
.unwrap();
|
||||
|
||||
// The whole bracket counts, retry included, matching the
|
||||
// in-process stopwatch.
|
||||
assert_eq!(stage(&state).live_inference_ms, 11_000);
|
||||
assert_eq!(stage(&state).inference, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stage_completion_replaces_the_live_estimate_with_finalized_timing() {
|
||||
let mut state = started_state();
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(3, "2026-04-07T12:00:09Z", agent_message()))
|
||||
.unwrap();
|
||||
assert_eq!(stage(&state).live_inference_ms, 9_000);
|
||||
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
4,
|
||||
"2026-04-07T12:00:10Z",
|
||||
EventBody::StageCompleted(completed_props(10_000, StageOutcome::Succeeded)),
|
||||
stage_id(),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let stage = stage(&state);
|
||||
assert_eq!(
|
||||
stage.live_timing(test_dt("2026-04-07T12:30:00Z")),
|
||||
stage.timing.unwrap(),
|
||||
"a terminal stage reports its finalized breakdown, not a live estimate"
|
||||
);
|
||||
assert_eq!(stage.live_inference_ms, 0);
|
||||
assert_eq!(stage.live_tool_ms, 0);
|
||||
assert!(stage.inference.is_none());
|
||||
assert!(stage.tool_batch.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_failure_freezes_open_work_and_clears_live_bookkeeping() {
|
||||
let mut state = started_state();
|
||||
state.status = RunStatus::Running;
|
||||
state
|
||||
.apply_event(&agent_event(2, "2026-04-07T12:00:01Z", llm_started()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(3, "2026-04-07T12:00:04Z", agent_message()))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&agent_event(
|
||||
4,
|
||||
"2026-04-07T12:00:05Z",
|
||||
tool_started("call-a"),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let mut failed = test_event(
|
||||
5,
|
||||
EventBody::RunFailed(run_failed_props(FailureReason::WorkflowError)),
|
||||
None,
|
||||
);
|
||||
failed.event.ts = test_dt("2026-04-07T12:00:10Z");
|
||||
state.apply_event(&failed).unwrap();
|
||||
|
||||
let stage = stage(&state);
|
||||
assert_eq!(
|
||||
stage.timing,
|
||||
Some(fabro_types::StageTiming::new(10_000, 3_000, 5_000))
|
||||
);
|
||||
assert_eq!(stage.state, StageState::Failed);
|
||||
assert_eq!(stage.live_inference_ms, 0);
|
||||
assert_eq!(stage.live_tool_ms, 0);
|
||||
assert!(stage.inference.is_none());
|
||||
assert!(stage.tool_batch.is_none());
|
||||
}
|
||||
}
|
||||
|
||||
fn test_event(seq: u32, body: EventBody, node_id: Option<&str>) -> EventEnvelope {
|
||||
let event = RunEvent {
|
||||
id: format!("evt-{seq}"),
|
||||
|
|
@ -2365,6 +2934,49 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parallel_branch_started_uses_the_graph_handler_for_live_timing() {
|
||||
for (handler_type, handler, expected) in [
|
||||
(
|
||||
"prompt",
|
||||
StageHandler::Prompt,
|
||||
StageTiming::new(5_000, 5_000, 0),
|
||||
),
|
||||
(
|
||||
"command",
|
||||
StageHandler::Command,
|
||||
StageTiming::new(5_000, 0, 5_000),
|
||||
),
|
||||
] {
|
||||
let mut spec = test_run_spec();
|
||||
let mut node = Node::new("review");
|
||||
node.attrs.insert(
|
||||
"type".to_string(),
|
||||
AttrValue::String(handler_type.to_string()),
|
||||
);
|
||||
spec.graph.nodes.insert(node.id.clone(), node);
|
||||
let mut state = RunProjection::new("Test run".to_string(), spec, Utc::now());
|
||||
let branch = StageId::new("review", 1);
|
||||
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
3,
|
||||
"2026-04-07T12:00:00Z",
|
||||
EventBody::ParallelBranchStarted(ParallelBranchStartedProps {
|
||||
index: 0,
|
||||
graph_visit: None,
|
||||
resumed_from_stage_id: None,
|
||||
}),
|
||||
branch.clone(),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let stage = state.stage(&branch).unwrap();
|
||||
assert_eq!(stage.handler, Some(handler));
|
||||
assert_eq!(stage.live_timing(test_dt("2026-04-07T12:00:05Z")), expected);
|
||||
}
|
||||
}
|
||||
|
||||
fn start_stage(state: &mut RunProjection, stage_id: &StageId) {
|
||||
state
|
||||
.apply_event(&test_stage_event(
|
||||
|
|
@ -2509,6 +3121,66 @@ mod tests {
|
|||
assert_eq!(provider_used.model.as_deref(), Some("fake"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn acp_events_accumulate_live_inference_time() {
|
||||
let mut state = initialized_projection();
|
||||
let stage_id = StageId::new("code", 1);
|
||||
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
3,
|
||||
"2026-04-07T12:00:00Z",
|
||||
EventBody::StageStarted(StageStartedProps {
|
||||
graph_visit: None,
|
||||
resumed_from_stage_id: None,
|
||||
index: 0,
|
||||
handler_type: "agent".to_string(),
|
||||
attempt: 1,
|
||||
max_attempts: 1,
|
||||
}),
|
||||
stage_id.clone(),
|
||||
))
|
||||
.unwrap();
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
4,
|
||||
"2026-04-07T12:00:05Z",
|
||||
EventBody::AgentAcpStarted(AgentAcpStartedProps {
|
||||
visit: 1,
|
||||
command: "python fake_agent.py".to_string(),
|
||||
config_name: Some("fake".to_string()),
|
||||
}),
|
||||
stage_id.clone(),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let stage = state.stage(&stage_id).unwrap();
|
||||
assert_eq!(
|
||||
stage.live_timing(test_dt("2026-04-07T12:00:15Z")),
|
||||
StageTiming::new(15_000, 10_000, 0)
|
||||
);
|
||||
|
||||
state
|
||||
.apply_event(&test_stage_event_at(
|
||||
5,
|
||||
"2026-04-07T12:00:17Z",
|
||||
EventBody::AgentAcpCompleted(AgentAcpCompletedProps {
|
||||
stdout: "done".to_string(),
|
||||
stderr: String::new(),
|
||||
stop_reason: "end_turn".to_string(),
|
||||
duration_ms: 12_000,
|
||||
}),
|
||||
stage_id.clone(),
|
||||
))
|
||||
.unwrap();
|
||||
|
||||
let stage = state.stage(&stage_id).unwrap();
|
||||
assert_eq!(
|
||||
stage.live_timing(test_dt("2026-04-07T12:00:20Z")),
|
||||
StageTiming::new(20_000, 12_000, 0)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn agent_acp_completed_updates_stage_output_projection() {
|
||||
let mut state = initialized_projection();
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ use std::fmt::Write as _;
|
|||
use std::sync::LazyLock;
|
||||
|
||||
use chrono::{DateTime, Utc};
|
||||
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming};
|
||||
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming, timing};
|
||||
use sqlx::sqlite::{SqliteConnection, SqliteRow};
|
||||
use sqlx::{QueryBuilder, Row as _, Sqlite, SqlitePool};
|
||||
use strum::VariantArray as _;
|
||||
|
|
@ -545,7 +545,7 @@ fn overlay_live_wall_time(run: &mut Run, now: DateTime<Utc>) {
|
|||
let Some(started_at) = run.timestamps.started_at else {
|
||||
return;
|
||||
};
|
||||
let wall_time_ms = RunTiming::wall_time_ms_since(started_at, now);
|
||||
let wall_time_ms = timing::elapsed_ms(started_at, now);
|
||||
run.timing = Some(
|
||||
run.timing
|
||||
.unwrap_or_else(|| RunTiming::wall_only(wall_time_ms))
|
||||
|
|
|
|||
|
|
@ -327,6 +327,18 @@ impl Database {
|
|||
Ok(self.projection_cache.get(run_id).await)
|
||||
}
|
||||
|
||||
pub async fn get_cached_projection(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
) -> Result<Option<Arc<RunProjection>>> {
|
||||
self.warm_projection_cache().await?;
|
||||
Ok(self
|
||||
.projection_cache
|
||||
.projection_snapshot(run_id)
|
||||
.await
|
||||
.map(|(projection, _)| projection))
|
||||
}
|
||||
|
||||
pub async fn get_cached_summary(
|
||||
&self,
|
||||
run_id: &RunId,
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
use std::collections::HashSet;
|
||||
|
||||
use fabro_graphviz::graph::Graph;
|
||||
|
||||
use crate::{Diagnostic, LintRule, Severity};
|
||||
|
|
@ -8,6 +10,26 @@ pub(super) fn rule() -> Box<dyn LintRule> {
|
|||
|
||||
struct Rule;
|
||||
|
||||
impl Rule {
|
||||
fn diagnostic(&self, node_id: &str, from: &str, to: &str) -> Diagnostic {
|
||||
Diagnostic {
|
||||
rule: self.name().to_string(),
|
||||
severity: Severity::Error,
|
||||
message: format!(
|
||||
"Node '{node_id}' is referenced by edge '{from} -> {to}' but has no node \
|
||||
declaration"
|
||||
),
|
||||
node_id: Some(node_id.to_string()),
|
||||
edge: Some((from.to_string(), to.to_string())),
|
||||
fix: Some(format!(
|
||||
"Declare node '{node_id}' or correct the edge endpoint"
|
||||
)),
|
||||
|
||||
..Diagnostic::default()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl LintRule for Rule {
|
||||
fn name(&self) -> &'static str {
|
||||
"edge_target_exists"
|
||||
|
|
@ -15,36 +37,12 @@ impl LintRule for Rule {
|
|||
|
||||
fn apply(&self, graph: &Graph) -> Vec<Diagnostic> {
|
||||
let mut diagnostics = Vec::new();
|
||||
let mut reported = HashSet::new();
|
||||
for edge in &graph.edges {
|
||||
if !graph.nodes.contains_key(&edge.to) {
|
||||
diagnostics.push(Diagnostic {
|
||||
rule: self.name().to_string(),
|
||||
severity: Severity::Error,
|
||||
message: format!(
|
||||
"Edge from '{}' targets non-existent node '{}'",
|
||||
edge.from, edge.to
|
||||
),
|
||||
node_id: None,
|
||||
edge: Some((edge.from.clone(), edge.to.clone())),
|
||||
fix: Some(format!("Define node '{}' or fix the edge target", edge.to)),
|
||||
|
||||
..Diagnostic::default()
|
||||
});
|
||||
}
|
||||
if !graph.nodes.contains_key(&edge.from) {
|
||||
diagnostics.push(Diagnostic {
|
||||
rule: self.name().to_string(),
|
||||
severity: Severity::Error,
|
||||
message: format!("Edge source '{}' references non-existent node", edge.from),
|
||||
node_id: None,
|
||||
edge: Some((edge.from.clone(), edge.to.clone())),
|
||||
fix: Some(format!(
|
||||
"Define node '{}' or fix the edge source",
|
||||
edge.from
|
||||
)),
|
||||
|
||||
..Diagnostic::default()
|
||||
});
|
||||
for endpoint in [&edge.to, &edge.from] {
|
||||
if !graph.nodes.contains_key(endpoint) && reported.insert(endpoint) {
|
||||
diagnostics.push(self.diagnostic(endpoint, &edge.from, &edge.to));
|
||||
}
|
||||
}
|
||||
}
|
||||
diagnostics
|
||||
|
|
@ -53,11 +51,112 @@ impl LintRule for Rule {
|
|||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use fabro_graphviz::graph::Edge;
|
||||
use fabro_graphviz::graph::{Edge, Graph};
|
||||
use fabro_graphviz::parser;
|
||||
|
||||
use super::Rule;
|
||||
use crate::rules::test_support::minimal_graph;
|
||||
use crate::{LintRule, Severity};
|
||||
use crate::{Diagnostic, LintRule, Severity};
|
||||
|
||||
fn parse(dot: &str) -> Graph {
|
||||
parser::parse(dot).expect("fixture should parse")
|
||||
}
|
||||
|
||||
fn undeclared_nodes(graph: &Graph) -> Vec<String> {
|
||||
Rule.apply(graph)
|
||||
.into_iter()
|
||||
.map(|d| d.node_id.expect("diagnostic should name a node"))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edge_only_node_is_rejected() {
|
||||
let graph = parse(
|
||||
r"digraph EdgeOnly {
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
start -> misspelled_node
|
||||
misspelled_node -> exit
|
||||
}",
|
||||
);
|
||||
|
||||
let diagnostics = Rule.apply(&graph);
|
||||
assert_eq!(diagnostics.len(), 1, "diagnostics: {diagnostics:?}");
|
||||
let Diagnostic {
|
||||
severity,
|
||||
node_id,
|
||||
edge,
|
||||
..
|
||||
} = &diagnostics[0];
|
||||
assert_eq!(*severity, Severity::Error);
|
||||
assert_eq!(node_id.as_deref(), Some("misspelled_node"));
|
||||
assert_eq!(
|
||||
edge.clone(),
|
||||
Some(("start".to_string(), "misspelled_node".to_string()))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn declaration_after_the_edge_is_accepted() {
|
||||
let graph = parse(
|
||||
r#"digraph DeclaredLater {
|
||||
start -> work
|
||||
work [prompt="Do the work"]
|
||||
work -> exit
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
}"#,
|
||||
);
|
||||
|
||||
assert!(Rule.apply(&graph).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chained_edges_report_every_undeclared_endpoint() {
|
||||
let graph = parse(
|
||||
r"digraph Chained {
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
start -> first -> second -> exit
|
||||
}",
|
||||
);
|
||||
|
||||
assert_eq!(undeclared_nodes(&graph), vec!["first", "second"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_node_is_reported_once_no_matter_how_many_edges_use_it() {
|
||||
let graph = parse(
|
||||
r"digraph Repeated {
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
start -> typo
|
||||
typo -> exit
|
||||
typo -> start
|
||||
}",
|
||||
);
|
||||
|
||||
assert_eq!(undeclared_nodes(&graph), vec!["typo"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subgraph_declaration_is_accepted() {
|
||||
let graph = parse(
|
||||
r#"digraph Subgraphed {
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
|
||||
subgraph cluster_loop {
|
||||
label = "Loop A"
|
||||
plan [prompt="Plan the work"]
|
||||
}
|
||||
|
||||
start -> plan -> exit
|
||||
}"#,
|
||||
);
|
||||
|
||||
assert!(Rule.apply(&graph).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edge_target_exists_rule_missing_target() {
|
||||
|
|
|
|||
|
|
@ -1,32 +1,7 @@
|
|||
use std::borrow::Cow;
|
||||
use std::collections::HashMap;
|
||||
|
||||
use fabro_model::Catalog;
|
||||
use fabro_types::{
|
||||
BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageTiming,
|
||||
};
|
||||
|
||||
fn stage_usage_with_cost<'a>(
|
||||
catalog: Option<&Catalog>,
|
||||
stage: &'a StageProjection,
|
||||
) -> Cow<'a, BilledTokenCounts> {
|
||||
let Some(catalog) = catalog else {
|
||||
return Cow::Borrowed(&stage.usage);
|
||||
};
|
||||
let Some(model) = stage.model.as_ref() else {
|
||||
return Cow::Borrowed(&stage.usage);
|
||||
};
|
||||
if stage.usage.total_usd_micros.is_some() {
|
||||
return Cow::Borrowed(&stage.usage);
|
||||
}
|
||||
|
||||
let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
|
||||
return Cow::Borrowed(&stage.usage);
|
||||
};
|
||||
let mut usage = stage.usage.clone();
|
||||
usage.total_usd_micros = Some(total_usd_micros);
|
||||
Cow::Owned(usage)
|
||||
}
|
||||
use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageTiming};
|
||||
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct ProjectionBillingStage {
|
||||
|
|
@ -80,7 +55,7 @@ pub fn billing_rollup_from_projection(
|
|||
if is_boundary_stage(projection, stage_id.node_id()) {
|
||||
continue;
|
||||
}
|
||||
let usage = stage_usage_with_cost(catalog, stage);
|
||||
let usage = stage.billed_usage(catalog);
|
||||
let usage = usage.as_ref();
|
||||
if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() {
|
||||
continue;
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ use crate::artifact_upload::ArtifactSink;
|
|||
use crate::condition::evaluate_condition;
|
||||
use crate::context::{Context, WorkflowContext, context_diff_public, keys};
|
||||
use crate::error::Error;
|
||||
use crate::operations::{ValidateInput, WorkflowInput, validate};
|
||||
use crate::operations::{ValidateInput, WorkflowInput, validate_with_catalog};
|
||||
use crate::outcome::{Outcome, OutcomeExt, StageOutcome};
|
||||
use crate::pipeline::types::Initialized;
|
||||
use crate::run_options::RunOptions;
|
||||
|
|
@ -65,20 +65,14 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
|
|||
.get("stack.child_dot_source")
|
||||
.and_then(|v| v.as_str())
|
||||
{
|
||||
let mut validated = validate(ValidateInput {
|
||||
workflow: WorkflowInput::DotSource {
|
||||
let graph = validate_child_workflow(
|
||||
WorkflowInput::DotSource {
|
||||
source: dot.to_string(),
|
||||
base_dir: None,
|
||||
},
|
||||
settings: WorkflowSettings::default(),
|
||||
vars: std::collections::HashMap::new(),
|
||||
cwd: cwd.clone(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: Arc::clone(&services.run.catalog),
|
||||
})?;
|
||||
validated.promote_template_undefined_variables_to_errors();
|
||||
validated.raise_on_errors()?;
|
||||
let (graph, _, _) = validated.into_parts();
|
||||
cwd,
|
||||
services,
|
||||
)?;
|
||||
return Ok(ParsedChildWorkflow {
|
||||
graph,
|
||||
workflow_path: None,
|
||||
|
|
@ -113,17 +107,7 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
|
|||
WorkflowInput::Bundled(workflow) => Some(workflow.path.clone()),
|
||||
WorkflowInput::Path(_) | WorkflowInput::DotSource { .. } => None,
|
||||
};
|
||||
let mut validated = validate(ValidateInput {
|
||||
workflow,
|
||||
settings: WorkflowSettings::default(),
|
||||
vars: std::collections::HashMap::new(),
|
||||
cwd,
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: Arc::clone(&services.run.catalog),
|
||||
})?;
|
||||
validated.promote_template_undefined_variables_to_errors();
|
||||
validated.raise_on_errors()?;
|
||||
let (graph, _, _) = validated.into_parts();
|
||||
let graph = validate_child_workflow(workflow, cwd, services)?;
|
||||
return Ok(ParsedChildWorkflow {
|
||||
graph,
|
||||
workflow_path,
|
||||
|
|
@ -132,6 +116,29 @@ fn parse_child_graph(node: &Node, services: &EngineServices) -> Result<ParsedChi
|
|||
Err(Error::handler("No child workflow source".to_string()))
|
||||
}
|
||||
|
||||
/// Validate a child workflow against the run's catalog, failing on any error
|
||||
/// diagnostic (undefined template variables included).
|
||||
fn validate_child_workflow(
|
||||
workflow: WorkflowInput,
|
||||
cwd: PathBuf,
|
||||
services: &EngineServices,
|
||||
) -> Result<Graph, Error> {
|
||||
let mut validated = validate_with_catalog(
|
||||
ValidateInput {
|
||||
workflow,
|
||||
settings: WorkflowSettings::default(),
|
||||
vars: HashMap::new(),
|
||||
cwd,
|
||||
custom_transforms: Vec::new(),
|
||||
},
|
||||
Arc::clone(&services.run.catalog),
|
||||
)?;
|
||||
validated.promote_template_undefined_variables_to_errors();
|
||||
validated.raise_on_errors()?;
|
||||
let (graph, _, _) = validated.into_parts();
|
||||
Ok(graph)
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl Handler for SubWorkflowHandler {
|
||||
async fn execute(
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ use crate::pipeline::{self, Persisted, TransformOptions, Validated};
|
|||
use crate::records::RunSpec;
|
||||
use crate::run_lookup::default_scratch_base;
|
||||
use crate::run_materialization::materialize_run;
|
||||
use crate::transforms::{RenderMode, Transform};
|
||||
use crate::transforms::{ModelResolutionTransform, RenderMode};
|
||||
use crate::workflow_bundle::{RunDefinition, WorkflowBundle};
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
|
|
@ -293,28 +293,21 @@ fn create_from_source(
|
|||
file_resolver: Option<Arc<dyn FileResolver>>,
|
||||
goal_override: Option<&str>,
|
||||
) -> Result<Persisted, Error> {
|
||||
let template_context = template_context(Some(&options.settings), vars);
|
||||
let mut validated = preprocess_and_validate(
|
||||
dot_source,
|
||||
options.source_name.clone(),
|
||||
let mut validated = preprocess_and_validate(dot_source, goal_override, &TransformOptions {
|
||||
current_dir,
|
||||
file_resolver,
|
||||
Vec::new(),
|
||||
template_context,
|
||||
goal_override,
|
||||
RenderMode::Structural,
|
||||
options
|
||||
.settings
|
||||
.run
|
||||
.model
|
||||
.provider
|
||||
.as_deref()
|
||||
.filter(|provider| !provider.is_empty())
|
||||
.map(ProviderId::new),
|
||||
&options.configured_providers,
|
||||
false,
|
||||
&options.catalog,
|
||||
)?;
|
||||
template_context: template_context(Some(&options.settings), vars),
|
||||
source_name: options.source_name.clone(),
|
||||
render_mode: RenderMode::Structural,
|
||||
custom_transforms: Vec::new(),
|
||||
model_resolution: Some(
|
||||
ModelResolutionTransform::for_eligible(
|
||||
Arc::clone(&options.catalog),
|
||||
options.configured_providers.iter().cloned().collect(),
|
||||
)
|
||||
.with_default_provider(configured_default_provider(&options.settings)),
|
||||
),
|
||||
})?;
|
||||
|
||||
validated.promote_template_undefined_variables_to_errors();
|
||||
if validated.has_errors() {
|
||||
|
|
@ -326,36 +319,37 @@ fn create_from_source(
|
|||
persist_validated(validated, options)
|
||||
}
|
||||
|
||||
/// Parse, transform, and validate `dot_source`.
|
||||
///
|
||||
/// `options.model_resolution` drives both halves of catalog awareness: it
|
||||
/// selects concrete models during TRANSFORM and enables the catalog-backed
|
||||
/// lint rules during VALIDATE. `None` leaves authored model and provider
|
||||
/// selectors untouched for offline structural validation.
|
||||
pub(super) fn preprocess_and_validate(
|
||||
dot_source: &str,
|
||||
source_name: Option<String>,
|
||||
current_dir: Option<PathBuf>,
|
||||
file_resolver: Option<Arc<dyn FileResolver>>,
|
||||
custom_transforms: Vec<Box<dyn Transform>>,
|
||||
template_context: TemplateContext,
|
||||
goal_override: Option<&str>,
|
||||
render_mode: RenderMode,
|
||||
default_provider: Option<ProviderId>,
|
||||
eligible_providers: &[ProviderId],
|
||||
catalog_fallback: bool,
|
||||
catalog: &Arc<Catalog>,
|
||||
options: &TransformOptions,
|
||||
) -> Result<Validated, Error> {
|
||||
let mut parsed = pipeline::parse(dot_source)?;
|
||||
apply_goal_override(&mut parsed.graph, goal_override);
|
||||
|
||||
let transformed = pipeline::transform(parsed, &TransformOptions {
|
||||
current_dir,
|
||||
file_resolver,
|
||||
template_context,
|
||||
source_name,
|
||||
render_mode,
|
||||
custom_transforms,
|
||||
catalog: Arc::clone(catalog),
|
||||
default_provider,
|
||||
eligible_providers: eligible_providers.iter().cloned().collect(),
|
||||
catalog_fallback,
|
||||
})?;
|
||||
Ok(pipeline::validate(transformed, catalog.as_ref(), &[]))
|
||||
let transformed = pipeline::transform(parsed, options)?;
|
||||
let catalog = options
|
||||
.model_resolution
|
||||
.as_ref()
|
||||
.map(ModelResolutionTransform::catalog);
|
||||
Ok(pipeline::validate(transformed, catalog, &[]))
|
||||
}
|
||||
|
||||
/// The workflow-level default provider, treating an empty setting as unset.
|
||||
pub(super) fn configured_default_provider(settings: &WorkflowSettings) -> Option<ProviderId> {
|
||||
settings
|
||||
.run
|
||||
.model
|
||||
.provider
|
||||
.as_deref()
|
||||
.filter(|provider| !provider.is_empty())
|
||||
.map(ProviderId::new)
|
||||
}
|
||||
|
||||
pub(super) fn template_context(
|
||||
|
|
@ -462,8 +456,9 @@ mod tests {
|
|||
use object_store::memory::InMemory;
|
||||
|
||||
use super::*;
|
||||
use crate::operations::{ValidateInput, validate};
|
||||
use crate::operations::{ValidateInput, validate, validate_with_catalog};
|
||||
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
|
||||
use crate::transforms::Transform;
|
||||
use crate::workflow_bundle::BundledWorkflow;
|
||||
fn memory_store() -> Arc<Database> {
|
||||
Arc::new(Database::new(
|
||||
|
|
@ -553,17 +548,19 @@ reasoning = false
|
|||
}
|
||||
|
||||
fn validate_dot(dot_source: &str, settings: WorkflowSettings) -> Validated {
|
||||
validate(ValidateInput {
|
||||
workflow: WorkflowInput::DotSource {
|
||||
source: dot_source.to_string(),
|
||||
base_dir: None,
|
||||
validate_with_catalog(
|
||||
ValidateInput {
|
||||
workflow: WorkflowInput::DotSource {
|
||||
source: dot_source.to_string(),
|
||||
base_dir: None,
|
||||
},
|
||||
settings,
|
||||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
},
|
||||
settings,
|
||||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
test_catalog(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
|
|
@ -573,21 +570,35 @@ reasoning = false
|
|||
fn validate_dot_with_vars(dot_source: &str, vars: HashMap<String, String>) -> Validated {
|
||||
preprocess_and_validate(
|
||||
dot_source,
|
||||
Some("workflow.fabro".to_string()),
|
||||
Some(PathBuf::from(".")),
|
||||
None,
|
||||
Vec::new(),
|
||||
template_context(Some(&WorkflowSettings::default()), vars),
|
||||
None,
|
||||
RenderMode::Structural,
|
||||
None,
|
||||
&test_provider_ids(),
|
||||
false,
|
||||
&test_catalog(),
|
||||
&test_transform_options(
|
||||
PathBuf::from("."),
|
||||
None,
|
||||
RenderMode::Structural,
|
||||
template_context(Some(&WorkflowSettings::default()), vars),
|
||||
),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Catalog-backed TRANSFORM options for the built-in test catalog.
|
||||
fn test_transform_options(
|
||||
current_dir: PathBuf,
|
||||
file_resolver: Option<Arc<dyn FileResolver>>,
|
||||
render_mode: RenderMode,
|
||||
template_context: TemplateContext,
|
||||
) -> TransformOptions {
|
||||
TransformOptions {
|
||||
current_dir: Some(current_dir),
|
||||
file_resolver,
|
||||
template_context,
|
||||
source_name: Some("workflow.fabro".to_string()),
|
||||
render_mode,
|
||||
custom_transforms: Vec::new(),
|
||||
model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
|
||||
}
|
||||
}
|
||||
|
||||
const MINIMAL_DOT: &str = r#"digraph Test {
|
||||
graph [goal="Build feature"]
|
||||
start [shape=Mdiamond]
|
||||
|
|
@ -744,17 +755,13 @@ reasoning = false
|
|||
|
||||
let result = preprocess_and_validate(
|
||||
dot,
|
||||
Some("workflow.fabro".to_string()),
|
||||
Some(PathBuf::from(".")),
|
||||
None,
|
||||
Vec::new(),
|
||||
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
|
||||
None,
|
||||
RenderMode::Strict,
|
||||
None,
|
||||
&test_provider_ids(),
|
||||
false,
|
||||
&test_catalog(),
|
||||
&test_transform_options(
|
||||
PathBuf::from("."),
|
||||
None,
|
||||
RenderMode::Strict,
|
||||
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
|
||||
),
|
||||
);
|
||||
let Err(err) = result else {
|
||||
panic!("expected strict mode to hard-fail on unbound inline prompt");
|
||||
|
|
@ -781,19 +788,15 @@ reasoning = false
|
|||
|
||||
let result = preprocess_and_validate(
|
||||
dot,
|
||||
Some("workflow.fabro".to_string()),
|
||||
Some(dir.path().to_path_buf()),
|
||||
Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
|
||||
None,
|
||||
))),
|
||||
Vec::new(),
|
||||
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
|
||||
None,
|
||||
RenderMode::Strict,
|
||||
None,
|
||||
&test_provider_ids(),
|
||||
false,
|
||||
&test_catalog(),
|
||||
&test_transform_options(
|
||||
dir.path().to_path_buf(),
|
||||
Some(Arc::new(crate::file_resolver::FilesystemFileResolver::new(
|
||||
None,
|
||||
))),
|
||||
RenderMode::Strict,
|
||||
template_context(Some(&WorkflowSettings::default()), HashMap::new()),
|
||||
),
|
||||
);
|
||||
let Err(err) = result else {
|
||||
panic!("expected strict mode to hard-fail on unbound imported prompt");
|
||||
|
|
@ -852,7 +855,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
});
|
||||
|
||||
assert!(result.is_err());
|
||||
|
|
@ -901,7 +903,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
let file_missing = validate(ValidateInput {
|
||||
|
|
@ -920,7 +921,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
|
|
@ -944,7 +944,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
let file_goal = validate(ValidateInput {
|
||||
|
|
@ -963,7 +962,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
|
|
@ -1055,7 +1053,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
});
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
|
@ -1100,7 +1097,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: vec![Box::new(TagTransform)],
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
validated.raise_on_errors().unwrap();
|
||||
|
|
@ -1135,7 +1131,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
validated.raise_on_errors().unwrap();
|
||||
|
|
@ -1177,7 +1172,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: dir.path().to_path_buf(),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
@ -1227,7 +1221,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
@ -1278,7 +1271,6 @@ reasoning = false
|
|||
vars: HashMap::new(),
|
||||
cwd: PathBuf::from("."),
|
||||
custom_transforms: Vec::new(),
|
||||
catalog: test_catalog(),
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ pub use rewind::{RewindInput, RewindOutcome, rewind};
|
|||
pub use source::WorkflowInput;
|
||||
pub use start::{StartServices, Started, start};
|
||||
pub use timeline::{ForkTarget, RunTimeline, TimelineEntry, build_timeline, timeline};
|
||||
pub use validate::{ValidateInput, validate, validate_with_ready_providers};
|
||||
pub use validate::{ValidateInput, validate, validate_with_catalog, validate_with_ready_providers};
|
||||
|
||||
pub use crate::pipeline::{LlmSpec, SandboxEnvSpec};
|
||||
pub use crate::transforms::RenderMode;
|
||||
|
|
|
|||
|
|
@ -5,12 +5,12 @@ use std::sync::Arc;
|
|||
use fabro_model::{Catalog, ProviderId};
|
||||
use fabro_types::WorkflowSettings;
|
||||
|
||||
use super::create::{preprocess_and_validate, template_context};
|
||||
use super::create::{configured_default_provider, preprocess_and_validate, template_context};
|
||||
use super::source::{ResolveWorkflowInput, WorkflowInput, resolve_workflow};
|
||||
use crate::error::Error;
|
||||
use crate::operations::RenderMode;
|
||||
use crate::pipeline::Validated;
|
||||
use crate::transforms::Transform;
|
||||
use crate::pipeline::{TransformOptions, Validated};
|
||||
use crate::transforms::{ModelResolutionTransform, Transform};
|
||||
|
||||
pub struct ValidateInput {
|
||||
pub workflow: WorkflowInput,
|
||||
|
|
@ -20,20 +20,24 @@ pub struct ValidateInput {
|
|||
pub vars: HashMap<String, String>,
|
||||
pub cwd: PathBuf,
|
||||
pub custom_transforms: Vec<Box<dyn Transform>>,
|
||||
pub catalog: Arc<Catalog>,
|
||||
}
|
||||
|
||||
/// Parse, transform, and validate a DOT source string.
|
||||
/// Parse, transform, and structurally validate a DOT source string without a
|
||||
/// model catalog. Model and provider availability is left to the caller that
|
||||
/// owns a catalog — typically the server.
|
||||
///
|
||||
/// Returns `Validated` even when validation produced errors. Call
|
||||
/// `validated.raise_on_errors()` if the caller wants to fail fast.
|
||||
pub fn validate(input: ValidateInput) -> Result<Validated, Error> {
|
||||
let eligible_providers = input
|
||||
.catalog
|
||||
.all_provider_ids()
|
||||
.into_iter()
|
||||
.collect::<Vec<_>>();
|
||||
validate_with_eligible_providers(input, &eligible_providers, false)
|
||||
validate_resolving_models(input, None)
|
||||
}
|
||||
|
||||
/// Parse, transform, and validate a DOT source string against `catalog`.
|
||||
pub fn validate_with_catalog(
|
||||
input: ValidateInput,
|
||||
catalog: Arc<Catalog>,
|
||||
) -> Result<Validated, Error> {
|
||||
validate_resolving_models(input, Some(ModelResolutionTransform::new(catalog)))
|
||||
}
|
||||
|
||||
/// Parse, transform, and validate, resolving models against the ready
|
||||
|
|
@ -41,45 +45,60 @@ pub fn validate(input: ValidateInput) -> Result<Validated, Error> {
|
|||
/// provider-readiness selection failures.
|
||||
pub fn validate_with_ready_providers(
|
||||
input: ValidateInput,
|
||||
catalog: Arc<Catalog>,
|
||||
ready_providers: &[ProviderId],
|
||||
) -> Result<Validated, Error> {
|
||||
validate_with_eligible_providers(input, ready_providers, true)
|
||||
validate_resolving_models(
|
||||
input,
|
||||
Some(
|
||||
ModelResolutionTransform::for_eligible(
|
||||
catalog,
|
||||
ready_providers.iter().cloned().collect(),
|
||||
)
|
||||
.with_catalog_fallback(true),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
fn validate_with_eligible_providers(
|
||||
/// The workflow's own default provider is only known once the workflow is
|
||||
/// resolved, so callers hand in a partially built transform and it is
|
||||
/// completed here.
|
||||
fn validate_resolving_models(
|
||||
input: ValidateInput,
|
||||
eligible_providers: &[ProviderId],
|
||||
catalog_fallback: bool,
|
||||
model_resolution: Option<ModelResolutionTransform>,
|
||||
) -> Result<Validated, Error> {
|
||||
let ValidateInput {
|
||||
workflow,
|
||||
settings,
|
||||
vars,
|
||||
cwd,
|
||||
custom_transforms,
|
||||
} = input;
|
||||
let resolved = resolve_workflow(ResolveWorkflowInput {
|
||||
workflow: input.workflow,
|
||||
settings: input.settings,
|
||||
cwd: input.cwd,
|
||||
workflow,
|
||||
settings,
|
||||
cwd,
|
||||
})
|
||||
.map_err(|err| Error::Parse(err.to_string()))?;
|
||||
|
||||
let model_resolution = model_resolution.map(|resolution| {
|
||||
resolution.with_default_provider(configured_default_provider(&resolved.settings))
|
||||
});
|
||||
|
||||
preprocess_and_validate(
|
||||
&resolved.raw_source,
|
||||
resolved
|
||||
.dot_path
|
||||
.as_ref()
|
||||
.map(|path| path.display().to_string()),
|
||||
resolved.current_dir,
|
||||
resolved.file_resolver,
|
||||
input.custom_transforms,
|
||||
template_context(Some(&resolved.settings), input.vars),
|
||||
resolved.goal_override.as_deref(),
|
||||
RenderMode::Structural,
|
||||
resolved
|
||||
.settings
|
||||
.run
|
||||
.model
|
||||
.provider
|
||||
.as_deref()
|
||||
.filter(|provider| !provider.is_empty())
|
||||
.map(fabro_model::ProviderId::new),
|
||||
eligible_providers,
|
||||
catalog_fallback,
|
||||
&input.catalog,
|
||||
&TransformOptions {
|
||||
current_dir: resolved.current_dir,
|
||||
file_resolver: resolved.file_resolver,
|
||||
template_context: template_context(Some(&resolved.settings), vars),
|
||||
source_name: resolved
|
||||
.dot_path
|
||||
.as_ref()
|
||||
.map(|path| path.display().to_string()),
|
||||
render_mode: RenderMode::Structural,
|
||||
custom_transforms,
|
||||
model_resolution,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,8 +3,8 @@ use std::sync::Arc;
|
|||
use super::types::{Parsed, TransformOptions, Transformed};
|
||||
use crate::error::Error;
|
||||
use crate::transforms::{
|
||||
FileInliningTransform, ImportTransform, ModelResolutionTransform,
|
||||
StylesheetApplicationTransform, TemplateTransform, Transform,
|
||||
FileInliningTransform, ImportTransform, StylesheetApplicationTransform, TemplateTransform,
|
||||
Transform,
|
||||
};
|
||||
|
||||
/// TRANSFORM phase: apply built-in and custom transforms to a parsed graph.
|
||||
|
|
@ -63,13 +63,10 @@ pub fn transform(parsed: Parsed, options: &TransformOptions) -> Result<Transform
|
|||
.apply_with_diagnostics(graph)?;
|
||||
diagnostics.extend(transform_diagnostics);
|
||||
let graph = StylesheetApplicationTransform.apply(graph)?;
|
||||
let graph = ModelResolutionTransform::for_eligible(
|
||||
Arc::clone(&options.catalog),
|
||||
options.eligible_providers.clone(),
|
||||
)
|
||||
.with_default_provider(options.default_provider.clone())
|
||||
.with_catalog_fallback(options.catalog_fallback)
|
||||
.apply(graph)?;
|
||||
let graph = match &options.model_resolution {
|
||||
Some(model_resolution) => model_resolution.apply(graph)?,
|
||||
None => graph,
|
||||
};
|
||||
|
||||
// Custom transforms
|
||||
let graph = options
|
||||
|
|
@ -98,6 +95,7 @@ mod tests {
|
|||
use crate::file_resolver::FilesystemFileResolver;
|
||||
use crate::pipeline::parse::parse;
|
||||
use crate::pipeline::types::{GOAL_SELF_REFERENCE_RULE, TEMPLATE_UNDEFINED_VARIABLE_RULE};
|
||||
use crate::transforms::ModelResolutionTransform;
|
||||
|
||||
fn write_file(path: &Path, contents: &str) {
|
||||
if let Some(parent) = path.parent() {
|
||||
|
|
@ -112,16 +110,13 @@ mod tests {
|
|||
|
||||
fn transform_options() -> TransformOptions {
|
||||
TransformOptions {
|
||||
current_dir: None,
|
||||
file_resolver: None,
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
catalog: test_catalog(),
|
||||
default_provider: None,
|
||||
eligible_providers: Catalog::builtin().all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
current_dir: None,
|
||||
file_resolver: None,
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
model_resolution: Some(ModelResolutionTransform::new(test_catalog())),
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -177,16 +172,9 @@ mod tests {
|
|||
)
|
||||
.unwrap();
|
||||
let transformed = transform(parsed, &TransformOptions {
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
catalog: test_catalog(),
|
||||
default_provider: None,
|
||||
eligible_providers: Catalog::builtin().all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
..transform_options()
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
@ -227,21 +215,15 @@ mod tests {
|
|||
)
|
||||
.unwrap();
|
||||
let transformed = transform(parsed, &TransformOptions {
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from(
|
||||
[(
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
template_context: fabro_template::TemplateContext::new().with_inputs(HashMap::from([
|
||||
(
|
||||
"task".to_string(),
|
||||
toml::Value::String("Launch".to_string()),
|
||||
)],
|
||||
)),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
catalog: test_catalog(),
|
||||
default_provider: None,
|
||||
eligible_providers: Catalog::builtin().all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
),
|
||||
])),
|
||||
..transform_options()
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
@ -349,6 +331,33 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structural_transform_preserves_catalog_owned_model_selection() {
|
||||
let dot = r#"digraph Test {
|
||||
graph [goal="Test"]
|
||||
start [shape=Mdiamond]
|
||||
work [prompt="Do work", model="private-model", provider="server-only"]
|
||||
exit [shape=Msquare]
|
||||
start -> work -> exit
|
||||
}"#;
|
||||
let parsed = parse(dot).unwrap();
|
||||
let transformed = transform(parsed, &TransformOptions {
|
||||
model_resolution: None,
|
||||
..transform_options()
|
||||
})
|
||||
.unwrap();
|
||||
let work = &transformed.graph.nodes["work"];
|
||||
|
||||
assert_eq!(
|
||||
work.attrs.get("model").and_then(AttrValue::as_str),
|
||||
Some("private-model")
|
||||
);
|
||||
assert_eq!(
|
||||
work.attrs.get("provider").and_then(AttrValue::as_str),
|
||||
Some("server-only")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transform_reports_goal_self_reference_once_across_passes() {
|
||||
// FileInlining renders the goal for prompt context, but TemplateTransform
|
||||
|
|
@ -365,16 +374,10 @@ mod tests {
|
|||
)
|
||||
.unwrap();
|
||||
let transformed = transform(parsed, &TransformOptions {
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Structural,
|
||||
custom_transforms: vec![],
|
||||
catalog: test_catalog(),
|
||||
default_provider: None,
|
||||
eligible_providers: Catalog::builtin().all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(Arc::new(FilesystemFileResolver::new(None))),
|
||||
render_mode: crate::operations::RenderMode::Structural,
|
||||
..transform_options()
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
use std::collections::{HashMap, HashSet};
|
||||
use std::collections::HashMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Arc;
|
||||
|
||||
|
|
@ -28,7 +28,7 @@ use crate::runtime_store::RunStoreHandle;
|
|||
use crate::services::{EngineServices, FabroRunToolServices, RunServices};
|
||||
use crate::stage_execution::StageExecutionSeed;
|
||||
use crate::steering_hub::SteeringHub;
|
||||
use crate::transforms::{RenderMode, Transform};
|
||||
use crate::transforms::{ModelResolutionTransform, RenderMode, Transform};
|
||||
use crate::workflow_bundle::WorkflowBundle;
|
||||
|
||||
/// Output of the PARSE phase.
|
||||
|
|
@ -359,18 +359,15 @@ pub struct Finalized {
|
|||
|
||||
/// Options for the TRANSFORM phase.
|
||||
pub struct TransformOptions {
|
||||
pub current_dir: Option<PathBuf>,
|
||||
pub file_resolver: Option<Arc<dyn FileResolver>>,
|
||||
pub template_context: TemplateContext,
|
||||
pub source_name: Option<String>,
|
||||
pub render_mode: RenderMode,
|
||||
pub custom_transforms: Vec<Box<dyn Transform>>,
|
||||
pub catalog: Arc<fabro_model::Catalog>,
|
||||
pub default_provider: Option<ProviderId>,
|
||||
pub eligible_providers: HashSet<ProviderId>,
|
||||
/// Fall back to the full catalog when the eligible providers cannot
|
||||
/// supply a requested model, instead of erroring.
|
||||
pub catalog_fallback: bool,
|
||||
pub current_dir: Option<PathBuf>,
|
||||
pub file_resolver: Option<Arc<dyn FileResolver>>,
|
||||
pub template_context: TemplateContext,
|
||||
pub source_name: Option<String>,
|
||||
pub render_mode: RenderMode,
|
||||
pub custom_transforms: Vec<Box<dyn Transform>>,
|
||||
/// Catalog-backed model resolution to perform. `None` preserves authored
|
||||
/// model and provider selectors for catalog-free structural validation.
|
||||
pub model_resolution: Option<ModelResolutionTransform>,
|
||||
}
|
||||
|
||||
/// Options for the FINALIZE phase.
|
||||
|
|
|
|||
|
|
@ -5,11 +5,15 @@ use super::types::{Transformed, Validated};
|
|||
|
||||
/// VALIDATE phase: run lint rules against the transformed graph.
|
||||
///
|
||||
/// Catalog-backed rules (model and provider availability) run only when
|
||||
/// `catalog` is `Some`. Offline callers pass `None` so a workflow naming a
|
||||
/// server-owned model is left for the server to judge.
|
||||
///
|
||||
/// **Infallible.** Always returns `Validated` with diagnostics. Caller decides
|
||||
/// whether to fail via `validated.raise_on_errors()`.
|
||||
pub fn validate(
|
||||
transformed: Transformed,
|
||||
catalog: &Catalog,
|
||||
catalog: Option<&Catalog>,
|
||||
extra_rules: &[&dyn LintRule],
|
||||
) -> Validated {
|
||||
let Transformed {
|
||||
|
|
@ -17,44 +21,33 @@ pub fn validate(
|
|||
source,
|
||||
mut diagnostics,
|
||||
} = transformed;
|
||||
diagnostics.extend(fabro_validate::validate_with_catalog(
|
||||
&graph,
|
||||
catalog,
|
||||
extra_rules,
|
||||
));
|
||||
diagnostics.extend(match catalog {
|
||||
Some(catalog) => fabro_validate::validate_with_catalog(&graph, catalog, extra_rules),
|
||||
None => fabro_validate::validate(&graph, extra_rules),
|
||||
});
|
||||
Validated::new(graph, source, diagnostics)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use fabro_model::Catalog;
|
||||
|
||||
use super::*;
|
||||
use crate::pipeline::parse::parse;
|
||||
use crate::pipeline::transform;
|
||||
use crate::pipeline::types::TransformOptions;
|
||||
|
||||
fn test_catalog() -> std::sync::Arc<Catalog> {
|
||||
std::sync::Arc::new(Catalog::from_builtin().unwrap())
|
||||
}
|
||||
|
||||
fn run_pipeline(dot: &str) -> Validated {
|
||||
let catalog = test_catalog();
|
||||
let parsed = parse(dot).unwrap();
|
||||
let transformed = transform::transform(parsed, &TransformOptions {
|
||||
current_dir: None,
|
||||
file_resolver: None,
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
catalog: std::sync::Arc::clone(&catalog),
|
||||
default_provider: None,
|
||||
eligible_providers: catalog.all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
current_dir: None,
|
||||
file_resolver: None,
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: crate::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
model_resolution: None,
|
||||
})
|
||||
.unwrap();
|
||||
validate(transformed, catalog.as_ref(), &[])
|
||||
validate(transformed, None, &[])
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
|
|
|||
|
|
@ -339,18 +339,18 @@ impl ImportTransform {
|
|||
.edges
|
||||
.retain(|edge| edge.from != placeholder_id && edge.to != placeholder_id);
|
||||
|
||||
for (node_id, node) in imported_graph.nodes {
|
||||
for (node_id, mut merged_node) in imported_graph.nodes {
|
||||
if node_id == start_id || node_id == exit_id {
|
||||
continue;
|
||||
}
|
||||
|
||||
let prefixed_id = format!("{placeholder_id}.{node_id}");
|
||||
let mut merged_node = Node::new(&prefixed_id);
|
||||
merged_node.id.clone_from(&prefixed_id);
|
||||
let imported_attrs = std::mem::take(&mut merged_node.attrs);
|
||||
merged_node.attrs.clone_from(&placeholder.default_attrs);
|
||||
merged_node.attrs.extend(node.attrs);
|
||||
merged_node.attrs.extend(imported_attrs);
|
||||
Self::remap_retry_target(&mut merged_node.attrs, placeholder_id);
|
||||
|
||||
merged_node.classes = node.classes;
|
||||
for class_name in &placeholder.class_names {
|
||||
Self::push_class(&mut merged_node.classes, class_name);
|
||||
}
|
||||
|
|
@ -649,7 +649,10 @@ impl PreparedImport {
|
|||
self.graph.nodes.iter().all(|(node_id, node)| {
|
||||
ImportTransform::is_start_sentinel(node_id, node)
|
||||
|| ImportTransform::is_exit_sentinel(node_id, node)
|
||||
})
|
||||
}) && matches!(
|
||||
self.graph.edges.as_slice(),
|
||||
[edge] if edge.from == self.start_id && edge.to == self.exit_id
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -909,6 +912,7 @@ mod tests {
|
|||
assert!(!graph.nodes.contains_key("validate"));
|
||||
assert!(graph.nodes.contains_key("validate.lint"));
|
||||
assert!(graph.nodes.contains_key("validate.test"));
|
||||
assert_eq!(graph.nodes["validate.lint"].id, "validate.lint");
|
||||
assert!(!graph.nodes.contains_key("validate.start"));
|
||||
assert!(!graph.nodes.contains_key("validate.exit"));
|
||||
|
||||
|
|
@ -951,6 +955,72 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edge_only_node_in_imported_fragment_stays_missing() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
write_file(
|
||||
&dir.path().join("validate.fabro"),
|
||||
r#"digraph validate {
|
||||
start [shape=Mdiamond]
|
||||
lint [prompt="Run clippy"]
|
||||
exit [shape=Msquare]
|
||||
start -> lint -> typo -> exit
|
||||
}"#,
|
||||
);
|
||||
|
||||
let graph = apply_import(
|
||||
r#"digraph Deploy {
|
||||
start [shape=Mdiamond]
|
||||
validate [import="./validate.fabro"]
|
||||
exit [shape=Msquare]
|
||||
start -> validate -> exit
|
||||
}"#,
|
||||
dir.path(),
|
||||
None,
|
||||
);
|
||||
|
||||
assert!(!graph.nodes.contains_key("validate.typo"));
|
||||
assert!(graph.nodes.contains_key("validate.lint"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edge_only_body_is_not_treated_as_empty_import() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
write_file(
|
||||
&dir.path().join("validate.fabro"),
|
||||
r"digraph validate {
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
start -> typo -> exit
|
||||
}",
|
||||
);
|
||||
|
||||
let graph = apply_import(
|
||||
r#"digraph Deploy {
|
||||
start [shape=Mdiamond]
|
||||
validate [import="./validate.fabro"]
|
||||
exit [shape=Msquare]
|
||||
start -> validate -> exit
|
||||
}"#,
|
||||
dir.path(),
|
||||
None,
|
||||
);
|
||||
|
||||
assert!(!graph.nodes.contains_key("validate.typo"));
|
||||
assert!(
|
||||
graph
|
||||
.edges
|
||||
.iter()
|
||||
.any(|edge| edge.from == "start" && edge.to == "validate.typo")
|
||||
);
|
||||
assert!(
|
||||
graph
|
||||
.edges
|
||||
.iter()
|
||||
.any(|edge| edge.from == "validate.typo" && edge.to == "exit")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn import_reports_structural_diagnostic_for_imported_prompt_templates() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
|
|
@ -1130,6 +1200,48 @@ mod tests {
|
|||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn imported_start_and_exit_sentinels_must_be_declared() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let host = r#"digraph Deploy {
|
||||
start [shape=Mdiamond]
|
||||
validate [import="./validate.fabro"]
|
||||
exit [shape=Msquare]
|
||||
start -> validate -> exit
|
||||
}"#;
|
||||
let cases = [
|
||||
(
|
||||
r#"digraph validate {
|
||||
work [prompt="Run checks"]
|
||||
exit [shape=Msquare]
|
||||
start -> work -> exit
|
||||
}"#,
|
||||
"imported workflow must have exactly one start node, found 0",
|
||||
),
|
||||
(
|
||||
r#"digraph validate {
|
||||
start [shape=Mdiamond]
|
||||
work [prompt="Run checks"]
|
||||
start -> work -> exit
|
||||
}"#,
|
||||
"imported workflow must have exactly one exit node, found 0",
|
||||
),
|
||||
];
|
||||
|
||||
for (source, expected_error) in cases {
|
||||
write_file(&dir.path().join("validate.fabro"), source);
|
||||
let graph = apply_import(host, dir.path(), None);
|
||||
|
||||
assert_eq!(
|
||||
graph.nodes["validate"]
|
||||
.attrs
|
||||
.get("import_error")
|
||||
.and_then(AttrValue::as_str),
|
||||
Some(expected_error)
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_entry_nodes_poison_placeholder() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
|
|
|
|||
|
|
@ -52,6 +52,13 @@ impl ModelResolutionTransform {
|
|||
self
|
||||
}
|
||||
|
||||
/// The catalog this transform resolves against, so callers can run the
|
||||
/// matching catalog-backed lint rules.
|
||||
#[must_use]
|
||||
pub fn catalog(&self) -> &Catalog {
|
||||
&self.catalog
|
||||
}
|
||||
|
||||
fn resolve_model(
|
||||
&self,
|
||||
model: &str,
|
||||
|
|
|
|||
|
|
@ -19,17 +19,18 @@ use crate::static_reference::{
|
|||
|
||||
/// How the template-expansion pass should treat undefined input variables.
|
||||
///
|
||||
/// Validate is structural — it should not fail just because the user has not
|
||||
/// bound `{{ inputs.* }}` yet. Run-start is strict — missing inputs are real
|
||||
/// errors. Splitting the two lets validate work on a bare `.fabro` while
|
||||
/// run-start preserves its current hard-fail behavior.
|
||||
/// Both validate and run-create render structurally so they can report every
|
||||
/// unbound `{{ inputs.* }}` variable in one pass rather than aborting on the
|
||||
/// first. Run-create then promotes the resulting warnings to errors, which
|
||||
/// keeps its hard-fail behavior.
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub enum RenderMode {
|
||||
/// Undefined inputs are hard errors. Used by run-create.
|
||||
/// Undefined inputs abort the pass with a hard error. No production caller
|
||||
/// uses this today; run-create promotes structural warnings instead.
|
||||
Strict,
|
||||
/// Undefined inputs render as empty and become warning diagnostics on the
|
||||
/// returned `Validated`, so structural lints still run. Used by
|
||||
/// `fabro validate`.
|
||||
/// `fabro validate` and by run-create.
|
||||
Structural,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -388,6 +388,9 @@ fn parse_and_validate_human_gate() {
|
|||
type="human"
|
||||
]
|
||||
|
||||
ship_it [prompt="Ship the change"]
|
||||
fixes [prompt="Apply the requested fixes"]
|
||||
|
||||
start -> review_gate
|
||||
review_gate -> ship_it [label="[A] Approve"]
|
||||
review_gate -> fixes [label="[F] Fix"]
|
||||
|
|
@ -4854,6 +4857,7 @@ async fn manager_loop_child_workflow_e2e() {
|
|||
#[tokio::test]
|
||||
async fn import_e2e_through_engine() {
|
||||
use fabro_workflow::pipeline::{TransformOptions, transform, validate};
|
||||
use fabro_workflow::transforms::ModelResolutionTransform;
|
||||
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let catalog = std::sync::Arc::new(
|
||||
|
|
@ -4897,21 +4901,20 @@ async fn import_e2e_through_engine() {
|
|||
)
|
||||
.expect("parse should succeed");
|
||||
let transformed = transform(parsed, &TransformOptions {
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(std::sync::Arc::new(
|
||||
current_dir: Some(dir.path().to_path_buf()),
|
||||
file_resolver: Some(std::sync::Arc::new(
|
||||
fabro_workflow::file_resolver::FilesystemFileResolver::new(None),
|
||||
)),
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: fabro_workflow::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
catalog: std::sync::Arc::clone(&catalog),
|
||||
default_provider: None,
|
||||
eligible_providers: catalog.all_provider_ids(),
|
||||
catalog_fallback: false,
|
||||
template_context: fabro_template::TemplateContext::new(),
|
||||
source_name: None,
|
||||
render_mode: fabro_workflow::operations::RenderMode::Strict,
|
||||
custom_transforms: vec![],
|
||||
model_resolution: Some(ModelResolutionTransform::new(std::sync::Arc::clone(
|
||||
&catalog,
|
||||
))),
|
||||
})
|
||||
.unwrap();
|
||||
let validated = validate(transformed, catalog.as_ref(), &[]);
|
||||
let validated = validate(transformed, Some(catalog.as_ref()), &[]);
|
||||
validated
|
||||
.raise_on_errors()
|
||||
.expect("validation should pass after imports expand");
|
||||
|
|
|
|||
|
|
@ -361,6 +361,11 @@ fn main() {
|
|||
"fabro_types::StageInferenceProjection",
|
||||
&[],
|
||||
),
|
||||
(
|
||||
"StageToolBatchProjection",
|
||||
"fabro_types::StageToolBatchProjection",
|
||||
&[],
|
||||
),
|
||||
("LlmOutputKind", "fabro_types::LlmOutputKind", &[]),
|
||||
("PermissionLevel", "fabro_types::PermissionLevel", &[]),
|
||||
(
|
||||
|
|
|
|||
|
|
@ -69,10 +69,10 @@ pub mod types {
|
|||
StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
|
||||
StageContextWindowStaleness, StageContextWindowUnavailableReason,
|
||||
StageContextWindowWarning, StageHandler, StageId, StageInferenceProjection,
|
||||
StageModelUsage, StageOutcome, StageProjection, StageState, SubAgentProjection,
|
||||
SubAgentStatus, SystemActorKind, SystemIntegrationStatus, SystemIntegrationsResponse,
|
||||
TodoListProjection, TurnId, UpdateVariableRequest, UserPrincipal, Variable,
|
||||
VariableListResponse, WorkflowSettings,
|
||||
StageModelUsage, StageOutcome, StageProjection, StageState, StageToolBatchProjection,
|
||||
SubAgentProjection, SubAgentStatus, SystemActorKind, SystemIntegrationStatus,
|
||||
SystemIntegrationsResponse, TodoListProjection, TurnId, UpdateVariableRequest,
|
||||
UserPrincipal, Variable, VariableListResponse, WorkflowSettings,
|
||||
};
|
||||
|
||||
pub use crate::generated::types::*;
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ use fabro_api::types::{
|
|||
StageContextWindowUnavailableReason as ApiStageContextWindowUnavailableReason,
|
||||
StageContextWindowWarning as ApiStageContextWindowWarning,
|
||||
StageInferenceProjection as ApiStageInferenceProjection, StageProjection as ApiStageProjection,
|
||||
StageToolBatchProjection as ApiStageToolBatchProjection,
|
||||
SubAgentProjection as ApiSubAgentProjection, SubAgentStatus as ApiSubAgentStatus,
|
||||
TodoListProjection as ApiTodoListProjection,
|
||||
};
|
||||
|
|
@ -29,8 +30,8 @@ use fabro_types::{
|
|||
ParallelBranchResult, PermissionLevel, SkillsProjection, StageContextWindow,
|
||||
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
|
||||
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
|
||||
StageContextWindowWarning, StageInferenceProjection, StageProjection, SubAgentProjection,
|
||||
SubAgentStatus, TodoListKind, TodoListProjection,
|
||||
StageContextWindowWarning, StageInferenceProjection, StageProjection, StageToolBatchProjection,
|
||||
SubAgentProjection, SubAgentStatus, TodoListKind, TodoListProjection,
|
||||
};
|
||||
use serde_json::json;
|
||||
|
||||
|
|
@ -68,9 +69,32 @@ fn stage_projection_reuses_nested_agent_state_types() {
|
|||
);
|
||||
assert_same_type::<ApiStageContextWindowWarning, StageContextWindowWarning>();
|
||||
assert_same_type::<ApiStageInferenceProjection, StageInferenceProjection>();
|
||||
assert_same_type::<ApiStageToolBatchProjection, StageToolBatchProjection>();
|
||||
assert_same_type::<ApiLlmOutputKind, LlmOutputKind>();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stage_tool_batch_projection_matches_openapi_json_shape() {
|
||||
let batch = StageToolBatchProjection {
|
||||
session_id: "ses_root".to_string(),
|
||||
started_at: "2026-04-29T12:34:00Z".parse().unwrap(),
|
||||
open_call_ids: ["call_1".to_string(), "call_2".to_string()]
|
||||
.into_iter()
|
||||
.collect(),
|
||||
};
|
||||
let value = serde_json::to_value(&batch).unwrap();
|
||||
assert_eq!(
|
||||
value,
|
||||
json!({
|
||||
"session_id": "ses_root",
|
||||
"started_at": "2026-04-29T12:34:00Z",
|
||||
"open_call_ids": ["call_1", "call_2"]
|
||||
})
|
||||
);
|
||||
let api_batch: ApiStageToolBatchProjection = serde_json::from_value(value).unwrap();
|
||||
assert_eq!(api_batch, batch);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stage_inference_projection_matches_openapi_json_shape() {
|
||||
let inference = StageInferenceProjection {
|
||||
|
|
@ -148,6 +172,10 @@ fn stage_projection_without_inference_round_trips() {
|
|||
|
||||
let stage: StageProjection = serde_json::from_value(value.clone()).unwrap();
|
||||
assert!(stage.inference.is_none());
|
||||
assert!(stage.acp_started_at.is_none());
|
||||
assert!(stage.tool_batch.is_none());
|
||||
assert_eq!(stage.live_inference_ms, 0);
|
||||
assert_eq!(stage.live_tool_ms, 0);
|
||||
assert_eq!(serde_json::to_value(stage).unwrap(), value);
|
||||
}
|
||||
|
||||
|
|
@ -311,6 +339,7 @@ fn stage_projection_round_trips_representative_json() {
|
|||
"first_output_kind": "text",
|
||||
"retries": 0
|
||||
},
|
||||
"acp_started_at": "2026-04-29T12:34:00Z",
|
||||
"agent_control": "running",
|
||||
"state": "succeeded"
|
||||
});
|
||||
|
|
|
|||
|
|
@ -125,7 +125,7 @@ pub use run_projection::{
|
|||
StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
|
||||
StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowUnavailableReason,
|
||||
StageContextWindowWarning, StageInferenceProjection, StageModelUsage, StageProjection,
|
||||
SubAgentProjection, SubAgentStatus, first_event_seq,
|
||||
StageToolBatchProjection, SubAgentProjection, SubAgentStatus, first_event_seq,
|
||||
};
|
||||
pub use run_sandbox::{
|
||||
RunSandbox, RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan,
|
||||
|
|
|
|||
|
|
@ -926,8 +926,6 @@ impl<'de> Deserialize<'de> for RunEvent {
|
|||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::HashMap;
|
||||
|
||||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
|
|
@ -992,20 +990,9 @@ mod tests {
|
|||
#[test]
|
||||
fn run_event_deserializes_adjacent_layout() {
|
||||
let settings = WorkflowSettings::default();
|
||||
let graph = Graph {
|
||||
name: "test".to_string(),
|
||||
nodes: HashMap::from([("start".to_string(), Node {
|
||||
id: "start".to_string(),
|
||||
attrs: HashMap::new(),
|
||||
classes: Vec::new(),
|
||||
})]),
|
||||
edges: vec![Edge {
|
||||
from: "start".to_string(),
|
||||
to: "done".to_string(),
|
||||
attrs: HashMap::new(),
|
||||
}],
|
||||
attrs: HashMap::new(),
|
||||
};
|
||||
let mut graph = Graph::new("test");
|
||||
graph.nodes.insert("start".to_string(), Node::new("start"));
|
||||
graph.edges.push(Edge::new("start", "done"));
|
||||
|
||||
let line = json!({
|
||||
"id": "evt_2",
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
use std::borrow::Cow;
|
||||
use std::collections::{BTreeMap, HashMap};
|
||||
use std::collections::{BTreeMap, BTreeSet, HashMap};
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use chrono::{DateTime, Utc};
|
||||
use fabro_model::{ReasoningEffort, Speed};
|
||||
use fabro_model::{Catalog, ReasoningEffort, Speed};
|
||||
use strum::{Display, EnumString, IntoStaticStr};
|
||||
|
||||
use crate::run_event::{AgentSessionActivatedProps, StagePromptProps};
|
||||
|
|
@ -12,7 +12,7 @@ use crate::{
|
|||
AgentToolSummary, BilledTokenCounts, Checkpoint, Conclusion, InterviewQuestionRecord,
|
||||
InvalidTransition, LlmOutputKind, ModelRef, PermissionLevel, PullRequestLink, RunApproval,
|
||||
RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, RunTiming, StageCompletion,
|
||||
StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection,
|
||||
StageHandler, StageId, StageState, StageTiming, StartRecord, TodoListProjection, timing,
|
||||
};
|
||||
|
||||
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
|
||||
|
|
@ -354,10 +354,31 @@ pub struct StageProjection {
|
|||
/// immutable projections under their own `StageId`s.
|
||||
///
|
||||
/// `None` for stages still in flight (`started_at` is set but no terminal
|
||||
/// event has been observed yet). For live wall-time ticking, the UI uses
|
||||
/// `started_at`; once terminal this carries the finalized breakdown.
|
||||
/// event has been observed yet). For a live breakdown while in flight, use
|
||||
/// [`StageProjection::live_timing`]; once terminal this carries the
|
||||
/// finalized, authoritative breakdown.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub timing: Option<StageTiming>,
|
||||
/// Inference time accumulated from closed brackets during this attempt.
|
||||
///
|
||||
/// Live estimate only: the authoritative value arrives with the terminal
|
||||
/// event and lands in `timing`. Excludes the currently-open bracket, which
|
||||
/// [`StageProjection::live_timing`] adds from `inference.started_at`.
|
||||
#[serde(default, skip_serializing_if = "is_zero_ms")]
|
||||
pub live_inference_ms: u64,
|
||||
/// Tool time accumulated from closed tool batches during this attempt.
|
||||
///
|
||||
/// A batch spans the first `agent.tool.started` with no outstanding calls
|
||||
/// through the `agent.tool.completed` that drains the last one, so tools
|
||||
/// running concurrently within a turn are counted once. This matches how
|
||||
/// the in-process stopwatch brackets `execute_tool_calls`; summing
|
||||
/// per-call durations would over-count parallel tool use.
|
||||
#[serde(default, skip_serializing_if = "is_zero_ms")]
|
||||
pub live_tool_ms: u64,
|
||||
/// Open tool batch for this stage: when the current batch started, and the
|
||||
/// `tool_call_id`s that have not yet reported completion.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub tool_batch: Option<StageToolBatchProjection>,
|
||||
#[serde(default)]
|
||||
pub usage: BilledTokenCounts,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
|
|
@ -390,11 +411,45 @@ pub struct StageProjection {
|
|||
/// the authority on whether a run is actually stuck.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub inference: Option<StageInferenceProjection>,
|
||||
/// Start of an external ACP agent process, if one is running.
|
||||
///
|
||||
/// ACP agents do not emit Fabro's internal LLM brackets, so their process
|
||||
/// lifetime is the best available live inference estimate.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub acp_started_at: Option<DateTime<Utc>>,
|
||||
#[serde(default)]
|
||||
pub agent_control: AgentControlState,
|
||||
pub state: StageState,
|
||||
}
|
||||
|
||||
/// Serde guard so zero-valued live accumulators stay off the wire.
|
||||
#[allow(
|
||||
clippy::trivially_copy_pass_by_ref,
|
||||
reason = "serde skip_serializing_if predicates receive fields by reference"
|
||||
)]
|
||||
fn is_zero_ms(value: &u64) -> bool {
|
||||
*value == 0
|
||||
}
|
||||
|
||||
/// One open tool batch: tool calls dispatched together that have not all
|
||||
/// reported completion.
|
||||
///
|
||||
/// `open_call_ids` is a set rather than a count because `agent.tool.completed`
|
||||
/// identifies its call by id, and a projection replaying a truncated or
|
||||
/// duplicated log must not let a repeated completion drain the batch early.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
|
||||
pub struct StageToolBatchProjection {
|
||||
/// Root agent session that dispatched the batch. Later transitions are
|
||||
/// gated on it so delayed events from a replaced session cannot mutate
|
||||
/// the current session's batch.
|
||||
pub session_id: String,
|
||||
/// When the batch opened — the first `agent.tool.started` observed while
|
||||
/// no other calls were outstanding.
|
||||
pub started_at: DateTime<Utc>,
|
||||
/// Calls dispatched but not yet completed, by `tool_call_id`.
|
||||
pub open_call_ids: BTreeSet<String>,
|
||||
}
|
||||
|
||||
/// One open inference bracket: a dispatched LLM request that has not yet
|
||||
/// produced a message, error, or interrupt.
|
||||
#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)]
|
||||
|
|
@ -506,6 +561,9 @@ impl StageProjection {
|
|||
response: None,
|
||||
completion: None,
|
||||
timing: None,
|
||||
live_inference_ms: 0,
|
||||
live_tool_ms: 0,
|
||||
tool_batch: None,
|
||||
usage: BilledTokenCounts::default(),
|
||||
model: None,
|
||||
root_agent_todos: None,
|
||||
|
|
@ -516,6 +574,7 @@ impl StageProjection {
|
|||
mcp_servers: Vec::new(),
|
||||
context_window: None,
|
||||
inference: None,
|
||||
acp_started_at: None,
|
||||
agent_control: AgentControlState::default(),
|
||||
provider_used: None,
|
||||
diff: None,
|
||||
|
|
@ -540,6 +599,29 @@ impl StageProjection {
|
|||
self.state
|
||||
}
|
||||
|
||||
/// This stage's token counts with a cost attached.
|
||||
///
|
||||
/// A provider-reported cost always wins. Otherwise the catalog prices the
|
||||
/// recorded tokens for the stage's model. The stored counts pass through
|
||||
/// untouched when there is no catalog, no model, or no price for that
|
||||
/// model. Empty usage also passes through untouched. These cases leave
|
||||
/// `total_usd_micros` as `None` rather than zero.
|
||||
#[must_use]
|
||||
pub fn billed_usage(&self, catalog: Option<&Catalog>) -> Cow<'_, BilledTokenCounts> {
|
||||
if self.usage.total_usd_micros.is_some() || self.usage.is_zero() {
|
||||
return Cow::Borrowed(&self.usage);
|
||||
}
|
||||
let (Some(catalog), Some(model)) = (catalog, self.model.as_ref()) else {
|
||||
return Cow::Borrowed(&self.usage);
|
||||
};
|
||||
let Some(total_usd_micros) = catalog.price_tokens(model, &self.usage.token_counts()) else {
|
||||
return Cow::Borrowed(&self.usage);
|
||||
};
|
||||
let mut usage = self.usage.clone();
|
||||
usage.total_usd_micros = Some(total_usd_micros);
|
||||
Cow::Owned(usage)
|
||||
}
|
||||
|
||||
/// Live wall-clock time in milliseconds.
|
||||
///
|
||||
/// While the stage is non-terminal (`Pending`, `Running`, or `Retrying`),
|
||||
|
|
@ -555,14 +637,197 @@ impl StageProjection {
|
|||
state,
|
||||
StageState::Running | StageState::Retrying | StageState::Pending
|
||||
) {
|
||||
return self.started_at.map(|started| {
|
||||
u64::try_from(now.signed_duration_since(started).num_milliseconds().max(0))
|
||||
.unwrap_or(0)
|
||||
});
|
||||
return self
|
||||
.started_at
|
||||
.map(|started| timing::elapsed_ms(started, now));
|
||||
}
|
||||
self.timing.map(|timing| timing.wall_time_ms)
|
||||
}
|
||||
|
||||
/// Live timing breakdown in milliseconds — the active-time twin of
|
||||
/// [`Self::live_wall_time_ms`].
|
||||
///
|
||||
/// Once terminal, returns the stored `timing` unchanged: the finalized
|
||||
/// breakdown comes from the worker's own stopwatch and is authoritative.
|
||||
///
|
||||
/// While in flight, returns an estimate reconstructed from the event log:
|
||||
/// accumulated closed brackets plus whatever bracket is open right now.
|
||||
/// The estimate is per-handler, because only agent stages emit brackets at
|
||||
/// all:
|
||||
///
|
||||
/// - `Agent` — accumulated inference and tool brackets, plus the open
|
||||
/// inference bracket and open tool batch.
|
||||
/// - `Prompt` — one inference call spanning the stage, so elapsed time
|
||||
/// since `started_at` counts as inference. Matches the finalized
|
||||
/// `active_only(inference, 0)`.
|
||||
/// - `Command` — the command *is* the work, so elapsed time counts as tool.
|
||||
/// Matches the finalized `active_only(0, duration_ms)`.
|
||||
/// - Everything else — zero. Waiting on a human, a timer, a condition, or
|
||||
/// child branches is wall time, not active time.
|
||||
///
|
||||
/// Active is clamped to wall. A worker killed mid-turn leaves its bracket
|
||||
/// open forever (see [`StageInferenceProjection`]), and without the clamp
|
||||
/// that bracket would tick up without bound. The clamp does not need to
|
||||
/// know the worker died: a stage cannot have been active longer than it
|
||||
/// has existed. `watchdog.timeout` remains the authority on whether a run
|
||||
/// is actually stuck.
|
||||
#[must_use]
|
||||
pub fn live_timing(&self, now: DateTime<Utc>) -> StageTiming {
|
||||
if let Some(timing) = self.timing {
|
||||
return timing;
|
||||
}
|
||||
|
||||
let wall_time_ms = self.live_wall_time_ms(now).unwrap_or(0);
|
||||
|
||||
// `handler` is absent on projections built from events written before
|
||||
// stage execution identity. Treat those as agent stages, matching
|
||||
// `StageHandler::from_handler_type`: the accumulators below are only
|
||||
// ever populated by agent events, so a legacy non-agent stage still
|
||||
// reads zero rather than being credited work it never did.
|
||||
let handler = self.handler.unwrap_or(StageHandler::Agent);
|
||||
let (inference_time_ms, tool_time_ms) = match handler {
|
||||
StageHandler::Agent => {
|
||||
let open_inference = self
|
||||
.inference
|
||||
.as_ref()
|
||||
.map(|inference| inference.started_at)
|
||||
.or(self.acp_started_at)
|
||||
.map_or(0, |started_at| timing::elapsed_ms(started_at, now));
|
||||
let open_tool = self
|
||||
.tool_batch
|
||||
.as_ref()
|
||||
.map_or(0, |batch| timing::elapsed_ms(batch.started_at, now));
|
||||
(
|
||||
self.live_inference_ms.saturating_add(open_inference),
|
||||
self.live_tool_ms.saturating_add(open_tool),
|
||||
)
|
||||
}
|
||||
StageHandler::Prompt => (wall_time_ms, 0),
|
||||
StageHandler::Command => (0, wall_time_ms),
|
||||
// Waiting on a human, a timer, a condition, or child branches is
|
||||
// wall time, not active time.
|
||||
StageHandler::Human
|
||||
| StageHandler::Wait
|
||||
| StageHandler::Conditional
|
||||
| StageHandler::Parallel
|
||||
| StageHandler::ParallelFanIn
|
||||
| StageHandler::StackManagerLoop
|
||||
| StageHandler::Start
|
||||
| StageHandler::Exit => (0, 0),
|
||||
};
|
||||
|
||||
StageTiming::new(wall_time_ms, inference_time_ms, tool_time_ms).clamped_to_wall()
|
||||
}
|
||||
|
||||
/// Fold a closed inference bracket into the live accumulator.
|
||||
pub fn accumulate_inference_ms(&mut self, elapsed_ms: u64) {
|
||||
self.live_inference_ms = self.live_inference_ms.saturating_add(elapsed_ms);
|
||||
}
|
||||
|
||||
/// Record the start of an external ACP agent process.
|
||||
pub fn open_acp_inference(&mut self, started_at: DateTime<Utc>) {
|
||||
self.close_open_acp_inference(started_at);
|
||||
self.acp_started_at = Some(started_at);
|
||||
}
|
||||
|
||||
/// Close an ACP process with its measured duration.
|
||||
///
|
||||
/// The duration covers the complete process. Use the larger value so a
|
||||
/// replayed terminal event or a projection restored mid-process cannot
|
||||
/// double-count it.
|
||||
pub fn close_acp_inference(&mut self, duration_ms: u64) {
|
||||
self.acp_started_at = None;
|
||||
self.live_inference_ms = self.live_inference_ms.max(duration_ms);
|
||||
}
|
||||
|
||||
/// Fold an open ACP process into the live accumulator at a run boundary.
|
||||
pub fn close_open_acp_inference(&mut self, now: DateTime<Utc>) {
|
||||
let Some(started_at) = self.acp_started_at.take() else {
|
||||
return;
|
||||
};
|
||||
self.accumulate_inference_ms(timing::elapsed_ms(started_at, now));
|
||||
}
|
||||
|
||||
/// Record a dispatched tool call, opening a batch if none is outstanding.
|
||||
///
|
||||
/// If a replacement root session starts work before the old session's end
|
||||
/// event arrives, freeze the old batch at this boundary before opening
|
||||
/// the new one. This keeps the sessions separate without dropping time.
|
||||
pub fn open_tool_call(
|
||||
&mut self,
|
||||
session_id: String,
|
||||
tool_call_id: String,
|
||||
started_at: DateTime<Utc>,
|
||||
) {
|
||||
let replaces_open_batch = self
|
||||
.tool_batch
|
||||
.as_ref()
|
||||
.is_some_and(|batch| batch.session_id != session_id);
|
||||
if replaces_open_batch {
|
||||
self.close_open_tool_batch(started_at);
|
||||
}
|
||||
self.tool_batch
|
||||
.get_or_insert_with(|| StageToolBatchProjection {
|
||||
session_id,
|
||||
started_at,
|
||||
open_call_ids: BTreeSet::new(),
|
||||
})
|
||||
.open_call_ids
|
||||
.insert(tool_call_id);
|
||||
}
|
||||
|
||||
/// Retire a tool call. Folds the batch into the live accumulator once the
|
||||
/// last outstanding call reports, so concurrent calls count once.
|
||||
pub fn close_tool_call(&mut self, session_id: &str, tool_call_id: &str, now: DateTime<Utc>) {
|
||||
let Some(batch) = self.tool_batch.as_mut() else {
|
||||
return;
|
||||
};
|
||||
if batch.session_id != session_id
|
||||
|| !batch.open_call_ids.remove(tool_call_id)
|
||||
|| !batch.open_call_ids.is_empty()
|
||||
{
|
||||
return;
|
||||
}
|
||||
self.close_open_tool_batch(now);
|
||||
}
|
||||
|
||||
/// Close a tool batch only when it belongs to `session_id`.
|
||||
pub fn close_tool_batch_for_session(&mut self, session_id: &str, now: DateTime<Utc>) {
|
||||
let opened_here = self
|
||||
.tool_batch
|
||||
.as_ref()
|
||||
.is_some_and(|batch| batch.session_id == session_id);
|
||||
if opened_here {
|
||||
self.close_open_tool_batch(now);
|
||||
}
|
||||
}
|
||||
|
||||
/// Fold any open tool batch into the live accumulator.
|
||||
pub fn close_open_tool_batch(&mut self, now: DateTime<Utc>) {
|
||||
let Some(batch) = self.tool_batch.take() else {
|
||||
return;
|
||||
};
|
||||
self.live_tool_ms = self
|
||||
.live_tool_ms
|
||||
.saturating_add(timing::elapsed_ms(batch.started_at, now));
|
||||
}
|
||||
|
||||
/// Install a worker-provided terminal timing and discard transient live
|
||||
/// bookkeeping that is no longer authoritative.
|
||||
pub fn set_authoritative_timing(&mut self, timing: StageTiming) {
|
||||
self.timing = Some(timing);
|
||||
self.clear_live_timing();
|
||||
}
|
||||
|
||||
/// Discard transient timing accumulators and open brackets.
|
||||
pub fn clear_live_timing(&mut self) {
|
||||
self.live_inference_ms = 0;
|
||||
self.live_tool_ms = 0;
|
||||
self.inference = None;
|
||||
self.acp_started_at = None;
|
||||
self.tool_batch = None;
|
||||
}
|
||||
|
||||
/// Begin a new automatic attempt within this stage execution: clear every
|
||||
/// per-attempt field so prior-attempt data does not leak, then record
|
||||
/// `started_at` and `state = Running`. Preserves `first_event_seq`
|
||||
|
|
@ -709,18 +974,21 @@ impl RunProjection {
|
|||
/// terminal conclusion yet.
|
||||
///
|
||||
/// Run-level wall time ticks from `run.started` to `now`. Active time sums
|
||||
/// inference and tool timing from stages that have already emitted a
|
||||
/// terminal stage event. Stage projections do not currently track live
|
||||
/// inference/tool time while a stage is still running, so active time steps
|
||||
/// forward when each stage completes while wall time advances continuously.
|
||||
/// [`StageProjection::live_timing`] across every stage, so an in-flight
|
||||
/// stage contributes its live estimate rather than nothing — both halves
|
||||
/// advance continuously. Terminal stages contribute their finalized,
|
||||
/// authoritative breakdown.
|
||||
///
|
||||
/// Active is not clamped to run wall time here: concurrent branches can
|
||||
/// legitimately sum past it. The clamp applies per stage.
|
||||
#[must_use]
|
||||
pub fn live_run_timing(&self, now: DateTime<Utc>) -> Option<RunTiming> {
|
||||
let start = self.start.as_ref()?;
|
||||
let wall_time_ms = RunTiming::wall_time_ms_since(start.start_time, now);
|
||||
let wall_time_ms = timing::elapsed_ms(start.start_time, now);
|
||||
let active = self
|
||||
.stages
|
||||
.values()
|
||||
.filter_map(|stage| stage.timing)
|
||||
.map(|stage| stage.live_timing(now))
|
||||
.fold(RunTiming::default(), |acc, timing| {
|
||||
acc.saturating_add(&RunTiming::from(timing))
|
||||
});
|
||||
|
|
@ -844,11 +1112,13 @@ mod iter_stages_tests {
|
|||
use std::num::NonZeroU32;
|
||||
|
||||
use chrono::Utc;
|
||||
use fabro_model::{Catalog, ModelRef, ProviderId};
|
||||
use serde_json::json;
|
||||
|
||||
use super::RunProjection;
|
||||
use crate::{
|
||||
AgentControlState, Graph, RunId, RunSpec, StageProjection, WorkflowSettings, test_support,
|
||||
AgentControlState, BilledTokenCounts, Graph, RunId, RunSpec, StageProjection,
|
||||
WorkflowSettings, test_support,
|
||||
};
|
||||
|
||||
fn seq(n: u32) -> NonZeroU32 {
|
||||
|
|
@ -970,4 +1240,361 @@ mod iter_stages_tests {
|
|||
assert_eq!(order, vec!["build@1", "verify@1", "verify@2"]);
|
||||
}
|
||||
}
|
||||
|
||||
fn priced_stage(total_usd_micros: Option<i64>) -> StageProjection {
|
||||
let mut stage = StageProjection::new(seq(1));
|
||||
stage.usage = BilledTokenCounts {
|
||||
input_tokens: 500_000,
|
||||
output_tokens: 125_000,
|
||||
total_tokens: 625_000,
|
||||
total_usd_micros,
|
||||
..BilledTokenCounts::default()
|
||||
};
|
||||
stage.model = Some(ModelRef {
|
||||
provider: ProviderId::openai(),
|
||||
model_id: "gpt-5.4".into(),
|
||||
speed: None,
|
||||
});
|
||||
stage
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn billed_usage_prices_uncosted_tokens_from_the_catalog() {
|
||||
let stage = priced_stage(None);
|
||||
|
||||
assert_eq!(stage.billed_usage(None).total_usd_micros, None);
|
||||
let priced = stage.billed_usage(Some(Catalog::builtin()));
|
||||
assert!(
|
||||
priced.total_usd_micros.is_some_and(|cost| cost > 0),
|
||||
"expected a catalog price, got {:?}",
|
||||
priced.total_usd_micros
|
||||
);
|
||||
// Pricing only fills in the cost; the token buckets pass through.
|
||||
assert_eq!(priced.input_tokens, 500_000);
|
||||
assert_eq!(priced.output_tokens, 125_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn billed_usage_keeps_a_provider_reported_cost_over_the_catalog_estimate() {
|
||||
let stage = priced_stage(Some(42));
|
||||
|
||||
assert_eq!(
|
||||
stage
|
||||
.billed_usage(Some(Catalog::builtin()))
|
||||
.total_usd_micros,
|
||||
Some(42)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn billed_usage_leaves_a_modelless_stage_uncosted() {
|
||||
let mut stage = priced_stage(None);
|
||||
stage.model = None;
|
||||
|
||||
assert_eq!(
|
||||
stage
|
||||
.billed_usage(Some(Catalog::builtin()))
|
||||
.total_usd_micros,
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn billed_usage_leaves_zero_tokens_uncosted() {
|
||||
let mut stage = priced_stage(None);
|
||||
stage.usage = BilledTokenCounts::default();
|
||||
|
||||
assert_eq!(
|
||||
stage
|
||||
.billed_usage(Some(Catalog::builtin()))
|
||||
.total_usd_micros,
|
||||
None
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod live_timing_tests {
|
||||
use std::collections::HashMap;
|
||||
|
||||
use chrono::{DateTime, TimeZone, Utc};
|
||||
|
||||
use super::{RunProjection, StageToolBatchProjection};
|
||||
use crate::{
|
||||
Graph, ModelRef, RunId, RunSpec, StageHandler, StageInferenceProjection, StageProjection,
|
||||
StageState, StageTiming, StartRecord, WorkflowSettings, first_event_seq, test_support,
|
||||
};
|
||||
|
||||
fn at(seconds: i64) -> DateTime<Utc> {
|
||||
Utc.timestamp_opt(1_700_000_000 + seconds, 0).unwrap()
|
||||
}
|
||||
|
||||
fn projection() -> RunProjection {
|
||||
RunProjection::new(
|
||||
"Test run".to_string(),
|
||||
RunSpec {
|
||||
run_id: RunId::new(),
|
||||
settings: WorkflowSettings::default(),
|
||||
graph: Graph::new("test"),
|
||||
graph_source: None,
|
||||
workflow_slug: None,
|
||||
automation: None,
|
||||
source_directory: None,
|
||||
labels: HashMap::default(),
|
||||
provenance: test_support::test_run_provenance(),
|
||||
manifest_blob: None,
|
||||
definition_blob: None,
|
||||
git: None,
|
||||
fork_source_ref: None,
|
||||
},
|
||||
at(0),
|
||||
)
|
||||
}
|
||||
|
||||
/// In-flight stage that started at `at(0)`.
|
||||
fn running(handler: StageHandler) -> StageProjection {
|
||||
let mut stage = StageProjection::new(first_event_seq(1));
|
||||
stage.handler = Some(handler);
|
||||
stage.started_at = Some(at(0));
|
||||
stage.state = StageState::Running;
|
||||
stage
|
||||
}
|
||||
|
||||
fn open_bracket(started_at: DateTime<Utc>) -> StageInferenceProjection {
|
||||
StageInferenceProjection {
|
||||
session_id: "session-1".to_string(),
|
||||
started_at,
|
||||
requested_model: ModelRef {
|
||||
provider: "anthropic".parse().unwrap(),
|
||||
model_id: "claude-sonnet-5".into(),
|
||||
speed: None,
|
||||
},
|
||||
first_output_at: None,
|
||||
first_output_kind: None,
|
||||
retries: 0,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_stage_returns_stored_timing_unchanged() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
stage.state = StageState::Succeeded;
|
||||
stage.timing = Some(StageTiming::new(90_000, 78_000, 7_000));
|
||||
// Live accumulators are stale leftovers; the finalized value wins.
|
||||
stage.live_inference_ms = 5;
|
||||
stage.live_tool_ms = 5;
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(600)),
|
||||
StageTiming::new(90_000, 78_000, 7_000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn agent_stage_sums_accumulators_and_open_brackets() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
stage.live_inference_ms = 30_000;
|
||||
stage.live_tool_ms = 5_000;
|
||||
// Inference open for 20s, tools open for 10s, at t=120s.
|
||||
stage.inference = Some(open_bracket(at(100)));
|
||||
stage.tool_batch = Some(StageToolBatchProjection {
|
||||
session_id: "session-1".to_string(),
|
||||
started_at: at(110),
|
||||
open_call_ids: ["call-1".to_string()].into_iter().collect(),
|
||||
});
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(120)),
|
||||
StageTiming::new(120_000, 50_000, 15_000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn agent_stage_without_brackets_reports_only_accumulators() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
stage.live_inference_ms = 30_000;
|
||||
stage.live_tool_ms = 5_000;
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(120)),
|
||||
StageTiming::new(120_000, 30_000, 5_000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn acp_process_counts_as_live_inference_and_uses_measured_duration() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
stage.open_acp_inference(at(30));
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(45)),
|
||||
StageTiming::new(45_000, 15_000, 0)
|
||||
);
|
||||
|
||||
stage.close_acp_inference(14_500);
|
||||
assert_eq!(
|
||||
stage.live_timing(at(45)),
|
||||
StageTiming::new(45_000, 14_500, 0)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prompt_stage_counts_elapsed_as_inference() {
|
||||
let stage = running(StageHandler::Prompt);
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(45)),
|
||||
StageTiming::new(45_000, 45_000, 0)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn command_stage_counts_elapsed_as_tool() {
|
||||
let stage = running(StageHandler::Command);
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(45)),
|
||||
StageTiming::new(45_000, 0, 45_000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn waiting_handlers_report_wall_time_with_zero_active() {
|
||||
for handler in [
|
||||
StageHandler::Human,
|
||||
StageHandler::Wait,
|
||||
StageHandler::Conditional,
|
||||
StageHandler::Parallel,
|
||||
StageHandler::ParallelFanIn,
|
||||
StageHandler::StackManagerLoop,
|
||||
StageHandler::Start,
|
||||
StageHandler::Exit,
|
||||
] {
|
||||
let stage = running(handler);
|
||||
let timing = stage.live_timing(at(600));
|
||||
|
||||
assert_eq!(
|
||||
timing,
|
||||
StageTiming::new(600_000, 0, 0),
|
||||
"{handler} should report wall time only"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn open_bracket_from_a_killed_worker_is_clamped_to_wall() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
// Bracket opened before the stage even started — the pathological
|
||||
// shape a killed worker leaves behind. Without the clamp this would
|
||||
// report 700s of inference against 600s of wall.
|
||||
stage.inference = Some(open_bracket(at(-100)));
|
||||
|
||||
let timing = stage.live_timing(at(600));
|
||||
|
||||
assert_eq!(timing.wall_time_ms, 600_000);
|
||||
assert_eq!(timing.active_time_ms, 600_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clamping_preserves_the_inference_tool_split() {
|
||||
let mut stage = running(StageHandler::Agent);
|
||||
// 3:1 inference:tool, totalling 200s of active against 100s of wall.
|
||||
stage.live_inference_ms = 150_000;
|
||||
stage.live_tool_ms = 50_000;
|
||||
|
||||
let timing = stage.live_timing(at(100));
|
||||
|
||||
assert_eq!(timing.wall_time_ms, 100_000);
|
||||
assert_eq!(timing.active_time_ms, 100_000);
|
||||
assert_eq!(timing.inference_time_ms, 75_000);
|
||||
assert_eq!(timing.tool_time_ms, 25_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn live_run_timing_counts_in_flight_stages_not_just_terminal_ones() {
|
||||
// The shape that motivated this change: two finished stages and one
|
||||
// long-running agent stage that had been active nearly the whole run.
|
||||
let mut projection = projection();
|
||||
projection.start = Some(StartRecord {
|
||||
start_time: at(0),
|
||||
run_branch: None,
|
||||
base_sha: None,
|
||||
});
|
||||
|
||||
let baseline = projection.stage_entry("baseline", 1, first_event_seq(1));
|
||||
baseline.handler = Some(StageHandler::Command);
|
||||
baseline.state = StageState::Succeeded;
|
||||
baseline.timing = Some(StageTiming::new(42_666, 0, 42_663));
|
||||
|
||||
let assess = projection.stage_entry("assess", 1, first_event_seq(2));
|
||||
assess.handler = Some(StageHandler::Agent);
|
||||
assess.state = StageState::Succeeded;
|
||||
assess.timing = Some(StageTiming::new(86_025, 78_230, 7_588));
|
||||
|
||||
let plan = projection.stage_entry("plan", 1, first_event_seq(3));
|
||||
plan.handler = Some(StageHandler::Agent);
|
||||
plan.started_at = Some(at(146));
|
||||
plan.state = StageState::Running;
|
||||
plan.live_inference_ms = 700_000;
|
||||
plan.live_tool_ms = 150_000;
|
||||
|
||||
let timing = projection.live_run_timing(at(1_013)).unwrap();
|
||||
|
||||
assert_eq!(timing.wall_time_ms, 1_013_000);
|
||||
assert_eq!(timing.inference_time_ms, 778_230);
|
||||
assert_eq!(timing.tool_time_ms, 200_251);
|
||||
// Before this change the in-flight stage contributed nothing and the
|
||||
// run reported 128,481 ms of active time against 1,013,000 ms of wall.
|
||||
assert_eq!(timing.active_time_ms, 978_481);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn live_run_timing_may_exceed_run_wall_when_branches_overlap() {
|
||||
let mut projection = projection();
|
||||
projection.start = Some(StartRecord {
|
||||
start_time: at(0),
|
||||
run_branch: None,
|
||||
base_sha: None,
|
||||
});
|
||||
|
||||
for (index, node) in ["branch-a", "branch-b", "branch-c"].iter().enumerate() {
|
||||
let stage =
|
||||
projection.stage_entry(node, 1, first_event_seq(u32::try_from(index).unwrap() + 1));
|
||||
stage.handler = Some(StageHandler::Agent);
|
||||
stage.state = StageState::Succeeded;
|
||||
stage.timing = Some(StageTiming::new(60_000, 60_000, 0));
|
||||
}
|
||||
|
||||
let timing = projection.live_run_timing(at(60)).unwrap();
|
||||
|
||||
assert_eq!(timing.wall_time_ms, 60_000);
|
||||
assert_eq!(
|
||||
timing.active_time_ms, 180_000,
|
||||
"concurrent branches legitimately sum past run wall time"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_legacy_stage_without_a_recorded_handler_uses_its_accumulators() {
|
||||
let mut stage = StageProjection::new(first_event_seq(1));
|
||||
stage.handler = None;
|
||||
stage.started_at = Some(at(0));
|
||||
stage.state = StageState::Running;
|
||||
stage.live_inference_ms = 30_000;
|
||||
|
||||
assert_eq!(
|
||||
stage.live_timing(at(120)),
|
||||
StageTiming::new(120_000, 30_000, 0)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_legacy_stage_with_no_accumulators_reports_no_active_time() {
|
||||
let mut stage = StageProjection::new(first_event_seq(1));
|
||||
stage.handler = None;
|
||||
stage.started_at = Some(at(0));
|
||||
stage.state = StageState::Running;
|
||||
|
||||
assert_eq!(stage.live_timing(at(120)), StageTiming::new(120_000, 0, 0));
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -18,6 +18,16 @@
|
|||
use chrono::{DateTime, Utc};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Non-negative milliseconds between two instants.
|
||||
///
|
||||
/// Clock skew or an out-of-order replay can put `end` before `start`; those
|
||||
/// spans contribute zero rather than wrapping into a large unsigned value.
|
||||
#[must_use]
|
||||
pub fn elapsed_ms(start: DateTime<Utc>, end: DateTime<Utc>) -> u64 {
|
||||
u64::try_from(end.signed_duration_since(start).num_milliseconds().max(0))
|
||||
.expect("non-negative chrono millisecond durations fit in u64")
|
||||
}
|
||||
|
||||
/// Timing breakdown for one stage visit.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct StageTiming {
|
||||
|
|
@ -62,6 +72,34 @@ impl StageTiming {
|
|||
Self::new(0, inference_time_ms, tool_time_ms)
|
||||
}
|
||||
|
||||
/// Scale the breakdown down so `active_time_ms` does not exceed
|
||||
/// `wall_time_ms`, preserving the inference/tool ratio.
|
||||
///
|
||||
/// Only meaningful for live estimates of a single in-flight stage, where
|
||||
/// an open bracket left behind by a killed worker would otherwise tick up
|
||||
/// without bound. Finalized timings come from the worker's stopwatch and
|
||||
/// already satisfy the invariant.
|
||||
///
|
||||
/// Deliberately *not* applied at run level: concurrent branches can
|
||||
/// legitimately sum past run wall time.
|
||||
#[must_use]
|
||||
pub fn clamped_to_wall(&self) -> Self {
|
||||
let active_time_ms = u128::from(self.inference_time_ms) + u128::from(self.tool_time_ms);
|
||||
if active_time_ms <= u128::from(self.wall_time_ms) {
|
||||
return *self;
|
||||
}
|
||||
// Preserve the split rather than truncating one side, so a clamped
|
||||
// stage still shows where its time went. Widen for the multiply: the
|
||||
// quotient is bounded by `wall_time_ms` because the exact, widened
|
||||
// active total exceeds it here, so it always fits back into u64.
|
||||
let scaled =
|
||||
u128::from(self.inference_time_ms) * u128::from(self.wall_time_ms) / active_time_ms;
|
||||
let inference_time_ms =
|
||||
u64::try_from(scaled).expect("scaled inference time is bounded by wall time");
|
||||
let tool_time_ms = self.wall_time_ms.saturating_sub(inference_time_ms);
|
||||
Self::new(self.wall_time_ms, inference_time_ms, tool_time_ms)
|
||||
}
|
||||
|
||||
/// Sum two timings field-by-field. Used to aggregate visits of one node
|
||||
/// and to accumulate run-level rollups.
|
||||
#[must_use]
|
||||
|
|
@ -136,13 +174,6 @@ impl RunTiming {
|
|||
..self
|
||||
}
|
||||
}
|
||||
|
||||
/// Milliseconds elapsed from `start` to `now`, clamped at zero.
|
||||
#[must_use]
|
||||
pub fn wall_time_ms_since(start: DateTime<Utc>, now: DateTime<Utc>) -> u64 {
|
||||
u64::try_from(now.signed_duration_since(start).num_milliseconds().max(0))
|
||||
.expect("non-negative milliseconds fit in u64")
|
||||
}
|
||||
}
|
||||
|
||||
impl From<StageTiming> for RunTiming {
|
||||
|
|
@ -158,7 +189,9 @@ impl From<StageTiming> for RunTiming {
|
|||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{RunTiming, StageTiming};
|
||||
use chrono::{TimeZone, Utc};
|
||||
|
||||
use super::{RunTiming, StageTiming, elapsed_ms};
|
||||
|
||||
#[test]
|
||||
fn stage_timing_new_derives_active_as_sum_of_inference_and_tool() {
|
||||
|
|
@ -189,6 +222,23 @@ mod tests {
|
|||
assert_eq!(sum.active_time_ms, 175);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stage_timing_clamp_uses_the_unsaturated_active_total() {
|
||||
let timing = StageTiming::new(u64::MAX, u64::MAX, u64::MAX).clamped_to_wall();
|
||||
|
||||
assert_eq!(timing.inference_time_ms, u64::MAX / 2);
|
||||
assert_eq!(timing.tool_time_ms, u64::MAX.saturating_sub(u64::MAX / 2));
|
||||
assert_eq!(timing.active_time_ms, u64::MAX);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn elapsed_ms_clamps_out_of_order_instants_to_zero() {
|
||||
let later = Utc.timestamp_opt(100, 0).unwrap();
|
||||
let earlier = Utc.timestamp_opt(99, 0).unwrap();
|
||||
|
||||
assert_eq!(elapsed_ms(later, earlier), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_timing_wall_only_zeroes_breakdown_and_active() {
|
||||
let timing = RunTiming::wall_only(1500);
|
||||
|
|
|
|||
|
|
@ -4,6 +4,10 @@
|
|||
// blank lines at end of file. Left alone, every regeneration produces a noisy
|
||||
// whitespace diff that masks real spec/client drift. This pass strips trailing
|
||||
// whitespace from every line and ends each file with exactly one newline.
|
||||
//
|
||||
// openapi-generator maps arrays with `uniqueItems` to `Set<T>`, but Axios
|
||||
// decodes and encodes their JSON representation as arrays. Keep the generated
|
||||
// type aligned with the runtime wire value.
|
||||
|
||||
import { Glob } from "bun";
|
||||
|
||||
|
|
@ -14,6 +18,7 @@ for await (const path of glob.scan(".")) {
|
|||
const original = await Bun.file(path).text();
|
||||
const normalized =
|
||||
original
|
||||
.replace(/\bSet</g, "Array<")
|
||||
.split("\n")
|
||||
.map((line) => line.replace(/\s+$/, ""))
|
||||
.join("\n")
|
||||
|
|
|
|||
|
|
@ -487,6 +487,7 @@ models/stage-projection.ts
|
|||
models/stage-state.ts
|
||||
models/stage-summary.ts
|
||||
models/stage-timing.ts
|
||||
models/stage-tool-batch-projection.ts
|
||||
models/start-record.ts
|
||||
models/start-run-request.ts
|
||||
models/steer-run-request.ts
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ export interface BatchDeleteRunsRequest {
|
|||
/**
|
||||
* Run IDs to process, in result order.
|
||||
*/
|
||||
'run_ids': Set<string>;
|
||||
'run_ids': Array<string>;
|
||||
/**
|
||||
* Whether to force deletion of active runs. Defaults to `false`.
|
||||
*/
|
||||
|
|
|
|||
|
|
@ -21,5 +21,5 @@ export interface BatchRunLifecycleRequest {
|
|||
/**
|
||||
* Run IDs to process, in result order.
|
||||
*/
|
||||
'run_ids': Set<string>;
|
||||
'run_ids': Array<string>;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -457,6 +457,7 @@ export * from './stage-projection';
|
|||
export * from './stage-state';
|
||||
export * from './stage-summary';
|
||||
export * from './stage-timing';
|
||||
export * from './stage-tool-batch-projection';
|
||||
export * from './start-record';
|
||||
export * from './start-run-request';
|
||||
export * from './steer-run-request';
|
||||
|
|
|
|||
|
|
@ -13,6 +13,9 @@
|
|||
*/
|
||||
|
||||
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
import type { BilledTokenCounts } from './billed-token-counts';
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
import type { StageHandler } from './stage-handler';
|
||||
|
|
@ -62,4 +65,8 @@ export interface RunStage {
|
|||
* Wall-clock time the latest attempt of this stage started, if known.
|
||||
*/
|
||||
'started_at'?: string | null;
|
||||
/**
|
||||
* Token counts for this stage execution alone. `total_usd_micros` is the provider-reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/billing` rows use. All-zero counts mean the stage made no model calls. Unlike the billing rows, which sum every visit of a node, this covers only this visit.
|
||||
*/
|
||||
'billing': BilledTokenCounts;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -15,12 +15,12 @@
|
|||
|
||||
|
||||
/**
|
||||
* Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently.
|
||||
* Timing rollup for an entire run. Active fields sum work across stage visits, so `active_time_ms` can exceed `wall_time_ms` when parallel branches run concurrently. For a running run, stages still in flight contribute a live estimate rather than nothing, so wall and active both advance continuously. Unlike `StageTiming`, active is not clamped to wall here — concurrent branches can legitimately sum past run wall time.
|
||||
*/
|
||||
export interface RunTiming {
|
||||
'wall_time_ms': number;
|
||||
'inference_time_ms'?: number;
|
||||
'tool_time_ms'?: number;
|
||||
'inference_time_ms': number;
|
||||
'tool_time_ms': number;
|
||||
/**
|
||||
* Equals `inference_time_ms + tool_time_ms`.
|
||||
*/
|
||||
|
|
|
|||
|
|
@ -60,6 +60,9 @@ import type { StageState } from './stage-state';
|
|||
import type { StageTiming } from './stage-timing';
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
import type { StageToolBatchProjection } from './stage-tool-batch-projection';
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
import type { SubAgentProjection } from './sub-agent-projection';
|
||||
// May contain unused imports in some cases
|
||||
// @ts-ignore
|
||||
|
|
@ -96,6 +99,15 @@ export interface StageProjection {
|
|||
*/
|
||||
'started_at'?: string | null;
|
||||
'timing'?: StageTiming | null;
|
||||
/**
|
||||
* Inference time accumulated from closed brackets during the current attempt. Live estimate only — the authoritative value arrives with the terminal event and lands in `timing`. Excludes the currently open bracket, whose span is measured from `inference.started_at`.
|
||||
*/
|
||||
'live_inference_ms'?: number;
|
||||
/**
|
||||
* Tool time accumulated from closed tool batches during the current attempt. A batch spans the first dispatched call through the completion that drains the last outstanding one, so tools running concurrently within a turn are counted once.
|
||||
*/
|
||||
'live_tool_ms'?: number;
|
||||
'tool_batch'?: StageToolBatchProjection | null;
|
||||
'usage': BilledTokenCounts;
|
||||
'model'?: BillingModelRef | null;
|
||||
'todos'?: TodoListProjection | null;
|
||||
|
|
@ -118,6 +130,10 @@ export interface StageProjection {
|
|||
'mcp_servers'?: Array<McpServerProjection>;
|
||||
'context_window'?: StageContextWindowProjection | null;
|
||||
'inference'?: StageInferenceProjection | null;
|
||||
/**
|
||||
* Start of an external ACP agent process, if one is running. ACP agents do not expose Fabro\'s internal LLM brackets, so the process lifetime supplies their live inference estimate.
|
||||
*/
|
||||
'acp_started_at'?: string | null;
|
||||
/**
|
||||
* Whether the agent is executing normally or waiting for steering after an interrupt.
|
||||
*/
|
||||
|
|
|
|||
|
|
@ -15,12 +15,12 @@
|
|||
|
||||
|
||||
/**
|
||||
* Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`.
|
||||
* Timing breakdown for one stage visit. Fields are all milliseconds. `wall_time_ms` is elapsed clock time; `inference_time_ms` is Fabro- observed LLM request/stream elapsed time; `tool_time_ms` is tool or command execution elapsed time; `active_time_ms` equals `inference_time_ms + tool_time_ms`. For a terminal stage these come from the worker\'s own stopwatch and are authoritative. For a stage still in flight they are a live estimate reconstructed from the event log, and `active_time_ms` is clamped to `wall_time_ms`. The estimate is replaced by the authoritative breakdown when the stage reaches a terminal event.
|
||||
*/
|
||||
export interface StageTiming {
|
||||
'wall_time_ms': number;
|
||||
'inference_time_ms'?: number;
|
||||
'tool_time_ms'?: number;
|
||||
'inference_time_ms': number;
|
||||
'tool_time_ms': number;
|
||||
/**
|
||||
* Equals `inference_time_ms + tool_time_ms`.
|
||||
*/
|
||||
|
|
|
|||
33
lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts
generated
Normal file
33
lib/packages/fabro-api-client/src/models/stage-tool-batch-projection.ts
generated
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
/* tslint:disable */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Fabro Run API
|
||||
* HTTP API for managing Fabro workflow run executions.
|
||||
*
|
||||
* The version of the OpenAPI document: 0.1.0
|
||||
*
|
||||
*
|
||||
* NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech).
|
||||
* https://openapi-generator.tech
|
||||
* Do not edit the class manually.
|
||||
*/
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* One open tool batch: tool calls dispatched together that have not all reported completion. `open_call_ids` is a set rather than a count so a duplicated completion in a replayed log cannot drain the batch early.
|
||||
*/
|
||||
export interface StageToolBatchProjection {
|
||||
/**
|
||||
* Root agent session that dispatched the batch. Transitions are gated on it so delayed events from a replaced session cannot mutate the current batch.
|
||||
*/
|
||||
'session_id': string;
|
||||
/**
|
||||
* When the batch opened — the first dispatched call observed while no other calls were outstanding.
|
||||
*/
|
||||
'started_at': string;
|
||||
/**
|
||||
* Calls dispatched but not yet completed, by tool call id.
|
||||
*/
|
||||
'open_call_ids': Array<string>;
|
||||
}
|
||||
10
test/edge_only_node.fabro
Normal file
10
test/edge_only_node.fabro
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
digraph EdgeOnlyNode {
|
||||
graph [goal="Reference a node that was never declared"]
|
||||
|
||||
/* `misspelled_node` is only ever named by an edge, never declared. */
|
||||
start [shape=Mdiamond, label="Start"]
|
||||
exit [shape=Msquare, label="Exit"]
|
||||
|
||||
start -> misspelled_node
|
||||
misspelled_node -> exit
|
||||
}
|
||||
10
test/server-model.fabro
Normal file
10
test/server-model.fabro
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
digraph ServerModel {
|
||||
graph [goal="Use a server-owned model"]
|
||||
|
||||
start [shape=Mdiamond, label="Start"]
|
||||
exit [shape=Msquare, label="Exit"]
|
||||
|
||||
work [label="Work", prompt="Do work", model="private-model", provider="server-only"]
|
||||
|
||||
start -> work -> exit
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue