Merge pull request #874 from fabro-sh/one-usage-type

One usage type: lithos-llm's Usage everywhere, and billing renamed to usage
This commit is contained in:
Bryan Helmkamp 2026-09-14 16:31:01 -04:00 • committed by GitHub
commit fa27cae5a3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
177 changed files with 3658 additions and 3700 deletions

8
Cargo.lock generated
View file

@ -4888,7 +4888,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77"
[[package]]
name = "lithos-llm"
version = "0.1.0"
source = "git+https://github.com/lithoscomputer/lithos-llm?rev=a1e3fd37b7153870411701327ac117606753fe90#a1e3fd37b7153870411701327ac117606753fe90"
source = "git+https://github.com/lithoscomputer/lithos-llm?rev=55add4596b861a0623d00c3a54aa5c147c8d504b#55add4596b861a0623d00c3a54aa5c147c8d504b"
dependencies = [
"async-trait",
"aws-config",
@ -5869,7 +5869,7 @@ dependencies = [
[[package]]
name = "pebble-agent"
version = "0.1.0"
source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5"
source = "git+https://github.com/lithoscomputer/pebble?rev=a39f43e26effdf99635eaf343f095c17157c9c93#a39f43e26effdf99635eaf343f095c17157c9c93"
dependencies = [
"async-trait",
"futures-util",
@ -5886,7 +5886,7 @@ dependencies = [
[[package]]
name = "pebble-cli-core"
version = "0.1.0"
source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5"
source = "git+https://github.com/lithoscomputer/pebble?rev=a39f43e26effdf99635eaf343f095c17157c9c93#a39f43e26effdf99635eaf343f095c17157c9c93"
dependencies = [
"anyhow",
"async-trait",
@ -5915,7 +5915,7 @@ dependencies = [
[[package]]
name = "pebble-coding-agent"
version = "0.1.0"
source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5"
source = "git+https://github.com/lithoscomputer/pebble?rev=a39f43e26effdf99635eaf343f095c17157c9c93#a39f43e26effdf99635eaf343f095c17157c9c93"
dependencies = [
"async-trait",
"futures-util",

View file

@ -93,7 +93,7 @@ insta = "1"
fabro-test = { path = "lib/foundation/fabro-test" }
# Provider-neutral LLM catalog and client. Pinned to a revision until 0.x is
# published to crates.io.
lithos-llm = { git = "https://github.com/lithoscomputer/lithos-llm", rev = "a1e3fd37b7153870411701327ac117606753fe90", default-features = false }
lithos-llm = { git = "https://github.com/lithoscomputer/lithos-llm", rev = "55add4596b861a0623d00c3a54aa5c147c8d504b", default-features = false }
# Deterministic OpenAI twin used by twin-mode E2E tests; the same revision
# lithos-llm verifies its codecs against.
twin-openai = { git = "https://github.com/lithoscomputer/twins", rev = "ca45f0e50a6716d716aa2f638ca3cf767e88f613" }
@ -122,9 +122,9 @@ sandbox-driver-testing = { git = "https://github.com/lithoscomputer/sandbox-driv
# sandbox, so the pebble and sandbox-driver pins move independently. Pebble
# pins the same lithos-llm rev as fabro, and its lockfile policy is that
# every shared crate resolves to the version lithos-llm locks.
pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5" }
pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5", features = ["mcp", "search-providers"] }
pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5" }
pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "a39f43e26effdf99635eaf343f095c17157c9c93" }
pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "a39f43e26effdf99635eaf343f095c17157c9c93", features = ["mcp", "search-providers"] }
pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "a39f43e26effdf99635eaf343f095c17157c9c93" }
sentry = { version = "0.35", default-features = false, features = ["backtrace", "contexts", "ureq", "rustls"] }
fork = "0.2"
exec = "0.3"

View file

@ -6,7 +6,7 @@ import {
RunSummaryPanelView,
type RunSummaryPanelViewProps,
} from "./run-summary-panel";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
function instanceText(instance: TestRenderer.ReactTestInstance): string {
const parts: string[] = [];
@ -56,7 +56,7 @@ function makeRun(overrides: Record<string, any> = {}) {
id: "run_1",
created_by: TEST_PRINCIPAL,
diff: null,
billing: null,
usage: makeUsage(),
...overrides,
} as any;
}
@ -163,9 +163,9 @@ describe("RunSummaryPanelView", () => {
);
});
test("renders cost from total_usd_micros", () => {
test("renders cost from the run's usage", () => {
const tree = render({
run: makeRun({ billing: { total_usd_micros: 840_000 } }),
run: makeRun({ usage: makeUsage({}, 840_000) }),
});
expect(instanceText(cellAfterLabel(tree, "Cost"))).toBe("$0.84");
});

View file

@ -128,7 +128,7 @@ export function RunSummaryPanelView({
artifactsLoading,
}: RunSummaryPanelViewProps) {
const diff = run?.diff ?? null;
const cost = formatUsdMicros(run?.billing?.total_usd_micros);
const cost = formatUsdMicros(run?.usage.cost?.usd_micros);
const sandboxKind = sandboxLifecycleKind(run?.sandbox);
return (

View file

@ -45,7 +45,7 @@ describe("SizeChip", () => {
.toBe("Size M · $12.34");
});
test("omits the cost when the run has no billing yet", () => {
test("omits the cost when the run has no cost yet", () => {
expect(tooltipLabel(<SizeChip size="M" />)).toBe("Size M");
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={null} />)).toBe("Size M");
});

View file

@ -21,36 +21,31 @@ import type {
McpToolSummary,
StageContextWindow,
StageProjection,
TokenUsage,
Usage,
} from "@qltysh/fabro-api-client";
import { StageInsightsSidebar } from "./stage-insights-sidebar";
const NO_USAGE: Usage = {
tokens: { input: 0, output: 0, reasoning: 0, cache_read: 0, cache_write: 0 },
};
function makeStage(overrides: Partial<StageProjection> = {}): StageProjection {
return {
first_event_seq: 1,
state: "running",
usage: {
input_tokens: 0,
output_tokens: 0,
cache_read_tokens: 0,
cache_create_tokens: 0,
total_tokens: 0,
} as StageProjection["usage"],
usage: NO_USAGE,
...overrides,
};
}
const NO_TOKENS: TokenUsage = { input: 0, output: 0, reasoning: 0, cache_read: 0, cache_write: 0 };
/** The coding agent's fold of a stage that has seen nothing yet. */
function makeAgent(overrides: Partial<AgentSessionProjection> = {}): AgentSessionProjection {
return {
root_session_id: "ses_root",
route: { provider: "anthropic", model: "claude-opus-4-7" },
activity: AgentSessionActivity.RUNNING,
usage: NO_TOKENS,
cost_usd_micros: null,
usage: NO_USAGE,
messages: 0,
descendants: {},
context_window: null,
@ -67,8 +62,7 @@ function makeAgent(overrides: Partial<AgentSessionProjection> = {}): AgentSessio
prompts: 1,
prompt: {
completed: false,
usage: NO_TOKENS,
cost_usd_micros: null,
usage: NO_USAGE,
messages: 0,
context_window: null,
tool_calls: 0,
@ -386,7 +380,7 @@ describe("StageInsightsSidebar", () => {
to: "openai/gpt-5.4",
attempt: 1,
error: agentError("rate limited"),
usage: NO_TOKENS,
usage: NO_USAGE,
inference_ms: 120,
tool_ms: 30,
continuation: FailoverContinuation.CONTINUE_TURN,

View file

@ -33,7 +33,7 @@ export function deriveStageSummary(events: EventEnvelope[]): StageSummary {
}
case "stage.completed": {
readFailure(summary, getObject(props, "failure"));
readBilling(summary, getObject(props, "billing"));
readUsage(summary, getObject(props, "usage"));
readTermination(summary, getObject(props, "termination"));
const notes = getString(props, "notes");
if (notes !== undefined) summary.notes = notes;
@ -43,7 +43,7 @@ export function deriveStageSummary(events: EventEnvelope[]): StageSummary {
}
case "stage.failed": {
readFailure(summary, getObject(props, "failure"));
readBilling(summary, getObject(props, "billing"));
readUsage(summary, getObject(props, "usage"));
break;
}
}
@ -59,10 +59,13 @@ function readFailure(summary: StageSummary, failure: unknown) {
if (actor !== undefined) summary.systemActor = actor;
}
function readBilling(summary: StageSummary, billing: unknown) {
if (!billing) return;
const input = getNumber(billing, "input_tokens");
const output = getNumber(billing, "output_tokens");
/** `stage.completed.usage` is a `ModelUsage`: the model, then the usage. */
function readUsage(summary: StageSummary, modelUsage: unknown) {
if (!modelUsage) return;
const tokens = getObject(getObject(modelUsage, "usage"), "tokens");
if (!tokens) return;
const input = getNumber(tokens, "input");
const output = getNumber(tokens, "output");
if (input !== undefined) summary.inputTokens = input;
if (output !== undefined) summary.outputTokens = output;
}

View file

@ -58,11 +58,16 @@ describe("deriveStageSummary", () => {
expect(summary.systemActor).toBe("agent");
});
test("captures billing tokens from stage.completed", () => {
test("captures usage tokens from stage.completed", () => {
const summary = deriveStageSummary([
makeEvent({
event: "stage.completed",
properties: { billing: { input_tokens: 12400, output_tokens: 3120 } },
properties: {
usage: {
model: { provider: "anthropic", model_id: "claude-sonnet-4-6" },
usage: { tokens: { input: 12400, output: 3120 } },
},
},
}),
]);
expect(summary.inputTokens).toBe(12400);
@ -118,7 +123,7 @@ describe("deriveStageSummary", () => {
test("tolerates missing or non-numeric properties", () => {
const summary = deriveStageSummary([
makeEvent({ event: "stage.started", properties: {} }),
makeEvent({ event: "stage.completed", properties: { billing: null } }),
makeEvent({ event: "stage.completed", properties: { usage: null } }),
]);
expect(summary).toEqual({});
});
@ -177,7 +182,10 @@ describe("StagePopover rendering", () => {
makeEvent({
event: "stage.completed",
properties: {
billing: { input_tokens: 12400, output_tokens: 3120 },
usage: {
model: { provider: "anthropic", model_id: "claude-sonnet-4-6" },
usage: { tokens: { input: 12400, output: 3120 } },
},
files_touched: ["a.rs", "b.rs"],
},
}),

View file

@ -3,7 +3,7 @@ import type { EventEnvelope } from "@qltysh/fabro-api-client";
import TestRenderer, { act } from "react-test-renderer";
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
import { makeUsage } from "../../lib/test-fixtures";
import type { Stage } from "../stage-sidebar";
import { FanInResults } from "./fan-in-results";
@ -23,7 +23,7 @@ const fanInStage: Stage = {
visit: 1,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
};
function event(seq: number, partial: Partial<EventEnvelope>): EventEnvelope {
@ -63,7 +63,7 @@ describe("FanInResults", () => {
event: "prompt.completed",
properties: {
response: "All branch findings are now available.",
billing: { input_tokens: 1200, output_tokens: 340 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 1200, output: 340 } } },
},
}),
]);

View file

@ -269,7 +269,7 @@ describe("parseReducerTranscript", () => {
event: "prompt.completed",
properties: {
response: "The branch results are joined.",
billing: { input_tokens: 1200, output_tokens: 340 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 1200, output: 340 } } },
},
}),
];

View file

@ -240,9 +240,10 @@ export function parseReducerTranscript(events: EventEnvelope[]): ReducerTranscri
} else if (event.event === "prompt.completed" && hasReducer) {
response = getString(props, "response") ?? response;
model = getString(props, "model") ?? model;
const billing = getObject(props, "billing") ?? {};
inputTokens = getNumber(billing, "input_tokens") ?? inputTokens;
outputTokens = getNumber(billing, "output_tokens") ?? outputTokens;
// `prompt.completed.usage` is a `ModelUsage`: the model, then the usage.
const tokens = getObject(getObject(getObject(props, "usage"), "usage"), "tokens") ?? {};
inputTokens = getNumber(tokens, "input") ?? inputTokens;
outputTokens = getNumber(tokens, "output") ?? outputTokens;
}
}

View file

@ -8,7 +8,7 @@ import {
mapRunToRunItem,
runStatusDisplay,
} from "./runs";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
function makeRun(overrides: Partial<Run> = {}): Run {
return {
@ -45,7 +45,7 @@ function makeRun(overrides: Partial<Run> = {}): Run {
tool_time_ms: 0,
active_time_ms: 0,
},
billing: { total_usd_micros: 500000 },
usage: makeUsage({}, 500000),
size: "XS",
diff: null,
pull_request: null,
@ -108,10 +108,10 @@ describe("mapRunListItem", () => {
expect(mapRunListItem(makeRun()).totalUsdMicros).toBe(500000);
});
test("leaves the billed total undefined for runs without terminal billing", () => {
expect(mapRunListItem(makeRun({ billing: null })).totalUsdMicros).toBeUndefined();
test("leaves the cost undefined for runs whose usage carries none", () => {
expect(mapRunListItem(makeRun({ usage: makeUsage() })).totalUsdMicros).toBeUndefined();
expect(
mapRunListItem(makeRun({ billing: { total_usd_micros: null } })).totalUsdMicros,
mapRunListItem(makeRun({ usage: makeUsage({ input: 12 }) })).totalUsdMicros,
).toBeUndefined();
});
});
@ -154,7 +154,7 @@ describe("mapRunToRunItem", () => {
completed_at: null,
},
timing: null,
billing: null,
usage: makeUsage(),
});
const item = mapRunToRunItem(summary);
expect(item.id).toBe("01DEF");

View file

@ -120,7 +120,7 @@ export function mapRunListItem(item: Run): RunItem {
additions: item.diff?.additions,
deletions: item.diff?.deletions,
size: item.size,
totalUsdMicros: item.billing?.total_usd_micros ?? undefined,
totalUsdMicros: item.usage.cost?.usd_micros,
};
}

View file

@ -1,28 +0,0 @@
import type { BilledTokenCounts } from "@qltysh/fabro-api-client";
export interface BillingTokenBucket {
label: string;
value: number;
}
export function billableOutputTokens(billing: BilledTokenCounts): number {
return billing.output_tokens + billing.reasoning_tokens;
}
/** The disjoint token buckets shown in every billing breakdown. */
export function billingTokenBuckets(billing: BilledTokenCounts): BillingTokenBucket[] {
return [
{ label: "Cache read", value: billing.cache_read_tokens },
{ label: "Cache creation", value: billing.cache_write_tokens },
{ label: "Uncached", value: billing.input_tokens },
{ label: "Output", value: billableOutputTokens(billing) },
];
}
export function hasBillingUsage(billing: BilledTokenCounts): boolean {
return (
billing.total_tokens !== 0 ||
(billing.total_usd_micros ?? 0) !== 0 ||
billingTokenBuckets(billing).some((bucket) => bucket.value !== 0)
);
}

View file

@ -122,7 +122,7 @@ function useLifecycleMutation(
// Keep the returned lifecycle state visible while revalidation
// observes the durable follow-up event (notably a 202 cancel).
void mutate(queryKeys.runs.detail(id), result.run, { revalidate: true });
void mutate(queryKeys.runs.billing(id));
void mutate(queryKeys.runs.usage(id));
}
mutateRunListCaches(mutate);
onSuccessExtra?.(result.run, mutate);

View file

@ -25,9 +25,9 @@ import type {
ProviderList,
PullRequestResponse,
RunArtifactListResponse,
RunBilling,
RunProjection,
Run,
RunUsage,
SandboxDetails,
SecretListResponse,
SandboxFileListResponse,
@ -286,10 +286,10 @@ export function useRunSettings<T = WorkflowSettings>(id: string | undefined) {
);
}
export function useRunBilling(id: string | undefined) {
return useSWR<RunBilling>(
id ? queryKeys.runs.billing(id) : null,
() => apiData(() => runOutputsApi.retrieveRunBilling(id!)),
export function useRunUsage(id: string | undefined) {
return useSWR<RunUsage>(
id ? queryKeys.runs.usage(id) : null,
() => apiData(() => runOutputsApi.retrieveRunUsage(id!)),
);
}

View file

@ -57,7 +57,7 @@ export const queryKeys = {
settings: (id: string) => ["runs", "settings", id] as const,
logs: (id: string) => ["runs", "logs", id] as const,
artifacts: (id: string) => ["runs", "artifacts", id] as const,
billing: (id: string) => ["runs", "billing", id] as const,
usage: (id: string) => ["runs", "usage", id] as const,
questions: (id: string, limit = 1, offset = 0) =>
["runs", "questions", id, limit, offset] as const,
events: (id: string, limit = 1000) => ["runs", "events", id, limit] as const,

View file

@ -29,7 +29,7 @@ import {
unarchiveRuns,
} from "./run-actions";
import { generatedAxios } from "./api-client";
import { TEST_PRINCIPAL } from "./test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "./test-fixtures";
type StubResponseInit = {
status: number;
@ -74,7 +74,7 @@ function makeRun(status: RunStatus, archived = false): Run {
last_event_at: null,
completed_at: null,
},
billing: null,
usage: makeUsage(),
size: "XS",
diff: null,
pull_request: null,

View file

@ -46,7 +46,7 @@ describe("queryKeysForRunEvent", () => {
queryKeys.runs.state("run-1"),
...queryKeys.runs.filesAllScopes("run-1"),
queryKeys.runs.commits("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.stages("run-1"),
queryKeys.runs.graph("run-1", "LR"),
queryKeys.runs.graph("run-1", "TB"),
@ -56,7 +56,7 @@ describe("queryKeysForRunEvent", () => {
test("stage.retrying invalidates stage-scoped and run-scoped resources", () => {
expect(queryKeysForRunEvent("run-1", "stage.retrying", "verify@2")).toEqual([
queryKeys.runs.stages("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.graph("run-1", "LR"),
queryKeys.runs.graph("run-1", "TB"),
@ -86,7 +86,7 @@ describe("queryKeysForRunEvent", () => {
test("interrupt settlement invalidates projected control state and stage activity", () => {
expect(queryKeysForRunEvent("run-1", "agent.round.interrupted", "nap@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.stageEvents("run-1", "nap@1"),
@ -166,7 +166,7 @@ describe("queryKeysForRunEvent", () => {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
@ -181,14 +181,14 @@ describe("queryKeysForRunEvent", () => {
).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
expect(queryKeysForRunEvent("run-1", "agent.session.ended")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
]);
});
@ -202,7 +202,7 @@ describe("queryKeysForRunEvent", () => {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
@ -213,7 +213,7 @@ describe("queryKeysForRunEvent", () => {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.usage("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
@ -285,7 +285,7 @@ describe("subscribeToRunEvents", () => {
source.emit({ event: "run.failed", run_id: "run-terminal" });
expect(source.closed).toBe(false);
expect(keys).toContainEqual(queryKeys.runs.files("run-terminal"));
expect(keys).toContainEqual(queryKeys.runs.billing("run-terminal"));
expect(keys).toContainEqual(queryKeys.runs.usage("run-terminal"));
keys.length = 0;
source.emit({ event: "run.archived", run_id: "run-terminal" });
@ -383,7 +383,7 @@ describe("subscribeToRunEvents", () => {
expect(source.closed).toBe(true);
expect(keys).toContainEqual(queryKeys.runs.files("run-terminal"));
expect(keys).toContainEqual(queryKeys.runs.billing("run-terminal"));
expect(keys).toContainEqual(queryKeys.runs.usage("run-terminal"));
cleanup();
coordinator.close();

View file

@ -177,7 +177,7 @@ function liveTimingKeys(runId: string): Key[] {
return [
queryKeys.runs.detail(runId),
queryKeys.runs.state(runId),
queryKeys.runs.billing(runId),
queryKeys.runs.usage(runId),
];
}
@ -199,7 +199,7 @@ export function queryKeysForRunEvent(
queryKeys.runs.state(runId),
...queryKeys.runs.filesAllScopes(runId),
queryKeys.runs.commits(runId),
queryKeys.runs.billing(runId),
queryKeys.runs.usage(runId),
queryKeys.runs.stages(runId),
queryKeys.runs.graph(runId, "LR"),
queryKeys.runs.graph(runId, "TB"),
@ -220,7 +220,7 @@ export function queryKeysForRunEvent(
if (STAGE_EVENTS.has(event)) {
const keys: Key[] = [
queryKeys.runs.stages(runId),
queryKeys.runs.billing(runId),
queryKeys.runs.usage(runId),
queryKeys.runs.events(runId, 1000),
queryKeys.runs.graph(runId, "LR"),
queryKeys.runs.graph(runId, "TB"),
@ -255,7 +255,7 @@ export function queryKeysForRunEvent(
if (event === "agent.round.interrupted") {
keys.unshift(
queryKeys.runs.detail(runId),
queryKeys.runs.billing(runId),
queryKeys.runs.usage(runId),
);
}
if (stageId) {
@ -372,7 +372,7 @@ function resyncKeysForRun(runId: string) {
queryKeys.runs.state(runId),
...queryKeys.runs.filesAllScopes(runId),
queryKeys.runs.commits(runId),
queryKeys.runs.billing(runId),
queryKeys.runs.usage(runId),
queryKeys.runs.stages(runId),
queryKeys.runs.events(runId, 1000),
queryKeys.runs.graph(runId, "LR"),

View file

@ -3,7 +3,7 @@ import type { PaginatedRunStageList, StageHandler, StageState } from "@qltysh/fa
import type { Stage } from "../components/stage-sidebar";
import { aggregateGraphNodeStatus, formatStageLabel, mapRunStagesToSidebarStages } from "./stage-sidebar";
import { makeBilledTokenCounts } from "./test-fixtures";
import { makeUsage } from "./test-fixtures";
import { makeStage as baseMakeStage } from "./test-utils";
function makeStage(nodeId: string, visit: number, status: StageState): Stage {
@ -28,15 +28,16 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
},
billing: makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
total_tokens: 43_690,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_usd_micros: 720_000,
}),
usage: makeUsage(
{
input: 28_640,
output: 7_550,
reasoning: 1_200,
cache_read: 4_800,
cache_write: 1_500,
},
720_000,
),
},
{
id: "apply-changes@2",
@ -45,7 +46,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "apply",
visit: 2,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },
@ -66,8 +67,8 @@ describe("mapRunStagesToSidebarStages", () => {
});
// Each visit keeps its own tokens and cost, so the stage popover never
// shows a sibling visit's usage.
expect(result[0].billing.total_usd_micros).toBe(720_000);
expect(result[1].billing.total_usd_micros).toBeUndefined();
expect(result[0].usage.cost?.usd_micros).toBe(720_000);
expect(result[1].usage.cost).toBeUndefined();
expect(formatStageLabel(result[0])).toBe("Apply Changes");
expect(result[1].id).toBe("apply-changes@2");
@ -87,7 +88,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "start",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
{
id: "verify@1",
@ -96,7 +97,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
{
id: "exit@1",
@ -105,7 +106,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "exit",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },
@ -125,7 +126,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },
@ -145,7 +146,7 @@ describe("mapRunStagesToSidebarStages", () => {
node_id: "work",
visit: 1,
graph_visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
{
id: "work@2",
@ -156,7 +157,7 @@ describe("mapRunStagesToSidebarStages", () => {
visit: 2,
graph_visit: 1,
resumed_from_stage_id: "work@1",
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },
@ -182,7 +183,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },
@ -225,7 +226,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "pending",
node_id: "approval",
visit: 1,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
},
],
meta: { has_more: false },

View file

@ -1,9 +1,9 @@
import { StageState } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
PaginatedRunStageList,
StageHandler,
StageModelUsage,
Usage,
} from "@qltysh/fabro-api-client";
import { isVisibleStage } from "../data/runs";
@ -33,10 +33,10 @@ export interface Stage {
startedAt: string | null;
providerUsed: StageModelUsage | null;
/**
* Tokens and cost for this visit alone, priced the same way the Billing tab
* Tokens and cost for this visit alone, priced the same way the Usage tab
* prices its per-node rows. All-zero counts mean the stage called no model.
*/
billing: BilledTokenCounts;
usage: Usage;
}
export const ACTIVE_STAGE_STATES: ReadonlySet<StageState> = new Set([
@ -114,7 +114,7 @@ export function mapRunStagesToSidebarStages(
: "--",
startedAt: stage.started_at ?? null,
providerUsed: stage.provider_used ?? null,
billing: stage.billing,
usage: stage.usage,
});
}
return stages;

View file

@ -1,4 +1,4 @@
import type { BilledTokenCounts, Principal } from "@qltysh/fabro-api-client";
import type { Cost, Principal, TokenCounts, Usage } from "@qltysh/fabro-api-client";
export const TEST_PRINCIPAL: Principal = {
kind: "user",
@ -7,16 +7,24 @@ export const TEST_PRINCIPAL: Principal = {
auth_method: "dev_token",
};
export function makeBilledTokenCounts(
overrides: Partial<BilledTokenCounts> = {},
): BilledTokenCounts {
export function makeTokenCounts(overrides: Partial<TokenCounts> = {}): TokenCounts {
return {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 0,
output_tokens: 0,
reasoning_tokens: 0,
total_tokens: 0,
input: 0,
output: 0,
reasoning: 0,
cache_read: 0,
cache_write: 0,
...overrides,
};
}
/** A usage with the given token buckets and, when `cost` is given, a catalog cost. */
export function makeUsage(
tokens: Partial<TokenCounts> = {},
cost?: number | Cost,
): Usage {
const usage: Usage = { tokens: makeTokenCounts(tokens) };
if (typeof cost === "number") usage.cost = { usd_micros: cost, source: "catalog" };
else if (cost) usage.cost = cost;
return usage;
}

View file

@ -3,7 +3,7 @@ import type { EventEnvelope } from "@qltysh/fabro-api-client";
import TestRenderer, { act } from "react-test-renderer";
import type { Stage } from "./stage-sidebar";
import { makeBilledTokenCounts } from "./test-fixtures";
import { makeUsage } from "./test-fixtures";
const IS_REACT_ACT_ENV = "IS_REACT_ACT_ENVIRONMENT" as const;
@ -85,7 +85,7 @@ export function makeStage(overrides: Partial<Stage> = {}): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
...overrides,
};
}

View file

@ -0,0 +1,53 @@
import type { Cost, TokenCounts, Usage } from "@qltysh/fabro-api-client";
export interface UsageTokenBucket {
label: string;
value: number;
}
/** The sum of the five disjoint token buckets. */
export function totalTokens(usage: Usage): number {
const { input, output, reasoning, cache_read, cache_write } = usage.tokens;
return input + output + reasoning + cache_read + cache_write;
}
/** Output tokens as they are priced: completion plus reasoning. */
export function billableOutputTokens(tokens: TokenCounts): number {
return tokens.output + tokens.reasoning;
}
/** The disjoint token buckets shown in every usage breakdown. */
export function usageTokenBuckets(usage: Usage): UsageTokenBucket[] {
return [
{ label: "Cache read", value: usage.tokens.cache_read },
{ label: "Cache creation", value: usage.tokens.cache_write },
{ label: "Uncached", value: usage.tokens.input },
{ label: "Output", value: billableOutputTokens(usage.tokens) },
];
}
/** Whether the usage carries any tokens or a cost. */
export function hasUsage(usage: Usage): boolean {
return totalTokens(usage) !== 0 || (usage.cost?.usd_micros ?? 0) !== 0;
}
/** The cost in USD micros, when the usage carries one. */
export function costUsdMicros(usage: Usage | null | undefined): number | undefined {
return usage?.cost?.usd_micros;
}
/**
* A short tag for a cost that did not come from the catalog: `reported`
* when the provider gave the figure, `summed` when it was assembled from
* differently sourced parts. Catalog estimates carry no tag.
*/
export function costSourceTag(cost: Cost | null | undefined): string | null {
switch (cost?.source) {
case "provider":
return "reported";
case "application":
return "summed";
default:
return null;
}
}

View file

@ -27,7 +27,7 @@ import * as RunChildren from "./routes/run-children";
import * as RunFiles from "./routes/run-files";
import * as RunSandbox from "./routes/run-sandbox";
import * as RunTerminal from "./routes/run-terminal";
import * as RunBilling from "./routes/run-billing";
import * as RunUsage from "./routes/run-usage";
import * as Insights from "./routes/insights";
import * as InsightsEditor from "./routes/insights-editor";
import * as InsightsNew from "./routes/insights-new";
@ -136,7 +136,7 @@ export const routes: RouteObject[] = [
route("files", RunFiles),
route("children", RunChildren),
route("sandbox", RunSandbox),
route("billing", RunBilling),
route("usage", RunUsage),
],
}),
route("insights", Insights, {

View file

@ -4,7 +4,7 @@ import TestRenderer, { act } from "react-test-renderer";
import { createMemoryRouter, RouterProvider } from "react-router";
import { ToastProvider } from "../components/toast";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
import { setupReactTestEnv } from "../lib/test-utils";
let currentRun: any = null;
@ -176,7 +176,7 @@ function makeRun(overrides: Record<string, unknown> = {}) {
completed_at: null,
},
timing: null,
billing: null,
usage: makeUsage(),
size: "XS",
ask_fabro: {
available: false,

View file

@ -14,7 +14,7 @@ import {
import { ToastProvider } from "../components/toast";
import { DemoModeProvider } from "../lib/demo-mode";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
let currentRunSummary: any = null;
let currentRunState: any = null;
@ -267,7 +267,7 @@ function makeRunSummary({
completed_at: null,
},
timing: null,
billing: null,
usage: makeUsage(),
size: "XS",
diff: diffSummary,
pull_request: pullRequest,

View file

@ -93,7 +93,7 @@ export function RunDetailHeader({
</span>
);
const sizeChip = (
<SizeChip size={summary.size} totalUsdMicros={summary.billing?.total_usd_micros} />
<SizeChip size={summary.size} totalUsdMicros={summary.usage.cost?.usd_micros} />
);
return (

View file

@ -16,7 +16,7 @@ const allTabs: RunDetailTabDefinition[] = [
{ name: "Files Changed", path: "/files", count: null },
{ name: "Children", path: "/children", count: null },
{ name: "Sandbox", path: "/sandbox", count: null, requiresSandbox: true },
{ name: "Billing", path: "/billing", count: null },
{ name: "Usage", path: "/usage", count: null },
];
export type RunDetailTab = RunDetailTabDefinition;

View file

@ -5,7 +5,7 @@ import { MemoryRouter, Route, Routes } from "react-router";
import { toast as sonnerToast } from "sonner";
import { ToastProvider } from "../components/toast";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
let currentFilesPayload: any = null;
let currentCommitsPayload: any = null;
@ -73,7 +73,7 @@ mock.module("../lib/queries", () => ({
last_event_at: null,
completed_at: null,
},
billing: null,
usage: makeUsage(),
size: "XS",
diff: null,
pull_request: null,

View file

@ -3,7 +3,7 @@ import { renderToStaticMarkup } from "react-dom/server";
import { StageState } from "@qltysh/fabro-api-client";
import type { Stage } from "../lib/stage-sidebar";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { makeUsage } from "../lib/test-fixtures";
import { StageChatView } from "./run-stages";
function stage(overrides: Partial<Stage> = {}): Stage {
@ -19,7 +19,7 @@ function stage(overrides: Partial<Stage> = {}): Stage {
resumedFromStageId: null,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
...overrides,
};
}

View file

@ -2,12 +2,12 @@ import { describe, expect, test } from "bun:test";
import { renderToStaticMarkup } from "react-dom/server";
import type {
BilledTokenCounts,
Usage,
ReasoningOutput,
StageModelUsage,
} from "@qltysh/fabro-api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { makeUsage } from "../lib/test-fixtures";
import { EventDetails, ModelUsagePopover } from "./run-stages";
const RUN_START = "2026-04-09T12:00:00Z";
@ -85,24 +85,25 @@ const PROVIDER_USED: StageModelUsage = {
reasoning_effort: "max",
};
function popoverMarkup(counts: BilledTokenCounts): string {
function popoverMarkup(usage: Usage): string {
return renderToStaticMarkup(
<ModelUsagePopover providerUsed={PROVIDER_USED} billing={counts} />,
<ModelUsagePopover providerUsed={PROVIDER_USED} usage={usage} />,
);
}
describe("ModelUsagePopover billing", () => {
describe("ModelUsagePopover usage", () => {
test("shows the visit's token buckets and cost next to the model", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_tokens: 43_690,
total_usd_micros: 720_000,
}),
makeUsage(
{
input: 28_640,
output: 7_550,
reasoning: 1_200,
cache_read: 4_800,
cache_write: 1_500,
},
720_000,
),
);
expect(html).toContain("kimi-k3");
@ -112,7 +113,7 @@ describe("ModelUsagePopover billing", () => {
expect(html).toContain("1.5k");
expect(html).toContain("Uncached");
expect(html).toContain("28.6k");
// Output folds in reasoning tokens, matching the Billing tab.
// Output folds in reasoning tokens, matching the Usage tab.
expect(html).toContain("Output");
expect(html).toContain("8.8k");
expect(html).toContain("Cost");
@ -120,7 +121,7 @@ describe("ModelUsagePopover billing", () => {
});
test("omits the token section for a stage that called no model", () => {
const html = popoverMarkup(makeBilledTokenCounts());
const html = popoverMarkup(makeUsage());
expect(html).toContain("kimi-k3");
expect(html).not.toContain("Tokens");
@ -129,11 +130,7 @@ describe("ModelUsagePopover billing", () => {
test("still shows tokens when nothing priced the stage", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 1_000,
output_tokens: 500,
total_tokens: 1_500,
}),
makeUsage({ input: 1_000, output: 500 }),
);
expect(html).toContain("Uncached");
@ -143,7 +140,7 @@ describe("ModelUsagePopover billing", () => {
test("shows a provider-reported cost when token counts are unavailable", () => {
const html = popoverMarkup(
makeBilledTokenCounts({ total_usd_micros: 720_000 }),
makeUsage({}, { usd_micros: 720_000, source: "provider" }),
);
expect(html).toContain("kimi-k3");

View file

@ -382,7 +382,7 @@ describe("eventsToActivity", () => {
response: "Refactored auth module",
model: "claude-sonnet-4-6",
provider: "anthropic",
billing: { input_tokens: 120, output_tokens: 30 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 120, output: 30 } } },
},
}),
];
@ -430,7 +430,7 @@ describe("eventsToActivity", () => {
response: "Done.",
model: "claude-sonnet-4-6",
provider: "anthropic",
billing: { input_tokens: 10, output_tokens: 5 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 10, output: 5 } } },
},
}),
];
@ -464,7 +464,7 @@ describe("eventsToActivity", () => {
response: "All clear.",
model: "claude-sonnet-4-6",
provider: "anthropic",
billing: { input_tokens: 0, output_tokens: 4 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 0, output: 4 } } },
},
}),
];
@ -1263,7 +1263,7 @@ describe("buildThreadDnaItems", () => {
});
describe("tool-call-only agent responses", () => {
test("retains an empty agent.message with its timestamp, billing, and tool-call count", () => {
test("retains an empty agent.message with its timestamp, usage, and tool-call count", () => {
const events: EventEnvelope[] = [
envelope(1, {
event: "agent.message",
@ -1307,7 +1307,7 @@ describe("tool-call-only agent responses", () => {
node_id: "code",
properties: {
response: "",
billing: { input_tokens: 1, output_tokens: 2 },
usage: { model: { provider: "anthropic", model_id: "claude-sonnet-4-6" }, usage: { tokens: { input: 1, output: 2 } } },
},
}),
];

View file

@ -69,7 +69,7 @@ import {
formatTokenCount,
formatUsdMicros,
} from "../lib/format";
import { billingTokenBuckets, hasBillingUsage } from "../lib/billing";
import { costSourceTag, hasUsage, usageTokenBuckets } from "../lib/usage";
import { plural } from "../lib/plural";
import {
useRun,
@ -96,11 +96,11 @@ import {
type UnknownRecord,
} from "../lib/unknown";
import type {
BilledTokenCounts,
EventEnvelope,
ReasoningOutput,
StageHandler,
StageModelUsage,
Usage,
} from "@qltysh/fabro-api-client";
export const handle = { wide: true, fullHeight: true };
@ -370,13 +370,15 @@ export function buildStageActivity(
}
case "prompt.completed": {
if (!sawAssistantMessage) {
const billing = (props.billing ?? {}) as UnknownRecord;
// `prompt.completed.usage` is a `ModelUsage`: the model, then the usage.
const tokens =
getObject(getObject(getObject(props, "usage"), "usage"), "tokens") ?? {};
turns.push({
kind: "assistant",
ts: e.ts,
content: getString(props, "response") ?? "",
inputTokens: getNumber(billing, "input_tokens") ?? 0,
outputTokens: getNumber(billing, "output_tokens") ?? 0,
inputTokens: getNumber(tokens, "input") ?? 0,
outputTokens: getNumber(tokens, "output") ?? 0,
toolCallCount: null,
// Only agent.message carries reasoning; prompt stages have none.
reasoning: null,
@ -887,10 +889,11 @@ export function formatStageModelUsageLabel(
const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
/** Tokens and cost for this stage visit alone. */
function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
if (!hasBillingUsage(billing)) return null;
const buckets = billingTokenBuckets(billing);
const cost = formatUsdMicros(billing.total_usd_micros);
function StageUsageRows({ usage }: { usage: Usage }) {
if (!hasUsage(usage)) return null;
const buckets = usageTokenBuckets(usage);
const cost = formatUsdMicros(usage.cost?.usd_micros);
const costTag = costSourceTag(usage.cost);
return (
<div className="mt-3">
<PopoverHeader>Tokens</PopoverHeader>
@ -905,7 +908,7 @@ function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
</PopoverRow>
))}
{cost && (
<PopoverRow label="Cost">
<PopoverRow label={costTag ? `Cost (${costTag})` : "Cost"}>
<span className={POPOVER_NUMBER}>{cost}</span>
</PopoverRow>
)}
@ -916,10 +919,10 @@ function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
export function ModelUsagePopover({
providerUsed,
billing,
usage,
}: {
providerUsed: StageModelUsage;
billing: BilledTokenCounts;
usage: Usage;
}) {
return (
<>
@ -942,7 +945,7 @@ export function ModelUsagePopover({
<PopoverRow label="Speed">{providerUsed.speed}</PopoverRow>
)}
</PopoverRows>
<StageBillingRows billing={billing} />
<StageUsageRows usage={usage} />
</>
);
}
@ -1956,7 +1959,7 @@ function EventsToolbar({
filteredCount,
totalCount,
providerUsed,
billing,
usage,
events,
runId,
stageId,
@ -1976,7 +1979,7 @@ function EventsToolbar({
filteredCount: number;
totalCount: number;
providerUsed: StageModelUsage | null;
billing: BilledTokenCounts;
usage: Usage;
events: EventEnvelope[];
runId: string;
stageId: string;
@ -2058,7 +2061,7 @@ function EventsToolbar({
showFilters ? "" : "ml-auto"
}`}
content={
<ModelUsagePopover providerUsed={providerUsed} billing={billing} />
<ModelUsagePopover providerUsed={providerUsed} usage={usage} />
}
>
<CpuChipIcon className="size-3.5" aria-hidden="true" />
@ -2414,7 +2417,7 @@ function RunStageActivityStage({
effectiveTab === "primary" ? turns.length : debugEvents.length
}
providerUsed={selectedStage.providerUsed}
billing={selectedStage.billing}
usage={selectedStage.usage}
events={stageEventsQuery.data ?? []}
runId={runId}
stageId={selectedStageId}

View file

@ -2,11 +2,11 @@ import { afterEach, describe, expect, mock, test } from "bun:test";
import TestRenderer from "react-test-renderer";
import type {
RunBilling,
RunUsage,
StageTiming,
} from "@qltysh/fabro-api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { makeUsage } from "../lib/test-fixtures";
function stageTiming(wall_time_ms = 0, inference_time_ms = 0, tool_time_ms = 0): StageTiming {
return {
@ -17,33 +17,33 @@ function stageTiming(wall_time_ms = 0, inference_time_ms = 0, tool_time_ms = 0):
};
}
let currentBilling: RunBilling | undefined;
let currentUsage: RunUsage | undefined;
mock.module("../lib/queries", () => ({
useRunBilling: () => ({ data: currentBilling }),
useRunUsage: () => ({ data: currentUsage }),
}));
const { default: RunBillingRoute } = await import("./run-billing");
const { default: RunUsageRoute } = await import("./run-usage");
function billing(overrides: Partial<RunBilling> = {}): RunBilling {
function runUsage(overrides: Partial<RunUsage> = {}): RunUsage {
return {
stages: [],
totals: {
timing: stageTiming(),
...makeBilledTokenCounts(),
usage: makeUsage(),
},
by_model: [],
...overrides,
};
}
function renderBilling(data: RunBilling): TestRenderer.ReactTestRenderer {
currentBilling = data;
function renderUsage(data: RunUsage): TestRenderer.ReactTestRenderer {
currentUsage = data;
(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true;
let renderer: TestRenderer.ReactTestRenderer | undefined;
TestRenderer.act(() => {
renderer = TestRenderer.create(<RunBillingRoute params={{ id: "run_1" }} />);
renderer = TestRenderer.create(<RunUsageRoute params={{ id: "run_1" }} />);
});
return renderer!;
}
@ -61,34 +61,34 @@ function textFromInstance(node: TestRenderer.ReactTestInstance): string {
.join("");
}
describe("RunBilling", () => {
describe("RunUsage", () => {
afterEach(() => {
currentBilling = undefined;
currentUsage = undefined;
delete (globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT;
});
test("shows a no-model-usage empty state when every stage is non-billable", () => {
const renderer = renderBilling(
billing({
test("shows a no-model-usage empty state when every stage called no model", () => {
const renderer = renderUsage(
runUsage({
stages: [
{
stage: { id: "start", name: "start" },
model: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
timing: stageTiming(),
state: "succeeded",
},
{
stage: { id: "command", name: "command" },
model: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
timing: stageTiming(61000),
state: "succeeded",
},
],
totals: {
timing: stageTiming(61000),
...makeBilledTokenCounts(),
usage: makeUsage(),
},
}),
);
@ -103,13 +103,13 @@ describe("RunBilling", () => {
});
test("renders mixed LLM and non-LLM rows while counting only LLM rows by model", () => {
const renderer = renderBilling(
billing({
const renderer = renderUsage(
runUsage({
stages: [
{
stage: { id: "start", name: "start" },
model: null,
billing: makeBilledTokenCounts(),
usage: makeUsage(),
timing: stageTiming(),
state: "succeeded",
},
@ -119,24 +119,14 @@ describe("RunBilling", () => {
provider: "anthropic",
model_id: "claude-sonnet-4-5",
},
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
timing: stageTiming(42000),
state: "succeeded",
},
],
totals: {
timing: stageTiming(42000),
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
},
by_model: [
{
@ -145,12 +135,7 @@ describe("RunBilling", () => {
model_id: "claude-sonnet-4-5",
},
stages: 1,
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
},
],
}),
@ -160,21 +145,98 @@ describe("RunBilling", () => {
expect(text).not.toContain("start");
expect(text).toContain("agent");
expect(text).toContain("By model");
expect(text).toContain("$0.24");
// A catalog estimate carries no source tag.
expect(text).not.toContain("reported");
const footers = renderer.root.findAll((node) => node.type === "tfoot");
const byModelFooterCells = footers[1].findAll((node) => node.type === "td");
expect(textFromInstance(byModelFooterCells[1])).toBe("1");
});
test("tags a provider-reported cost with its source", () => {
const reported = makeUsage({ input: 1200, output: 300 }, {
usd_micros: 240000,
source: "provider",
});
const renderer = renderUsage(
runUsage({
stages: [
{
stage: { id: "agent", name: "agent" },
model: { provider: "openrouter", model_id: "kimi-k3" },
usage: reported,
timing: stageTiming(42000),
state: "succeeded",
},
],
totals: { timing: stageTiming(42000), usage: reported },
by_model: [
{
model: { provider: "openrouter", model_id: "kimi-k3" },
stages: 1,
usage: reported,
},
],
}),
);
const text = textFromNode(renderer.toJSON());
expect(text).toContain("$0.24");
expect(text).toContain("reported");
});
test("says unknown for a total whose cost is unknown, not zero", () => {
// One stage priced from the catalog, one the catalog could not price:
// the rows keep their own costs and the total has none.
const priced = makeUsage({ input: 1200, output: 300 }, 240000);
const unpriced = makeUsage({ input: 500, output: 50 });
const total = makeUsage({ input: 1700, output: 350 });
const renderer = renderUsage(
runUsage({
stages: [
{
stage: { id: "plan", name: "plan" },
model: { provider: "openai", model_id: "gpt-5.4" },
usage: priced,
timing: stageTiming(1000),
state: "succeeded",
},
{
stage: { id: "work", name: "work" },
model: { provider: "openai", model_id: "mystery" },
usage: unpriced,
timing: stageTiming(2000),
state: "succeeded",
},
],
totals: { timing: stageTiming(3000), usage: total },
by_model: [
{ model: { provider: "openai", model_id: "gpt-5.4" }, stages: 1, usage: priced },
{ model: { provider: "openai", model_id: "mystery" }, stages: 1, usage: unpriced },
],
}),
);
const text = textFromNode(renderer.toJSON());
expect(text).toContain("$0.24");
expect(text).toContain("unknown");
expect(text).not.toContain("$0.00");
const footers = renderer.root.findAll((node) => node.type === "tfoot");
const footerCells = footers[0].findAll((node) => node.type === "td");
expect(textFromInstance(footerCells[4])).toBe("unknown");
});
test("keeps the empty state for runs with no stages", () => {
const renderer = renderBilling(billing());
const renderer = renderUsage(runUsage());
const text = textFromNode(renderer.toJSON());
expect(text).toContain("No stages yet");
expect(text).toContain("Stages will appear as soon as the run starts executing.");
});
test("renders an in-flight row with live billing and includes its elapsed time in the footer", () => {
test("renders an in-flight row with live usage and includes its elapsed time in the footer", () => {
const originalNow = Date.now;
// Pin "now" to 30s after the in-flight row started.
const startedAt = "2026-04-29T12:00:00.000Z";
@ -182,8 +244,8 @@ describe("RunBilling", () => {
Date.now = () => fakeNow;
try {
const renderer = renderBilling(
billing({
const renderer = renderUsage(
runUsage({
stages: [
{
stage: { id: "in-flight", name: "in-flight" },
@ -192,12 +254,7 @@ describe("RunBilling", () => {
model_id: "claude-opus-4-6",
speed: "fast",
},
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
timing: stageTiming(),
started_at: startedAt,
state: "running",
@ -205,12 +262,7 @@ describe("RunBilling", () => {
],
totals: {
timing: stageTiming(),
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
},
by_model: [
{
@ -220,12 +272,7 @@ describe("RunBilling", () => {
speed: "fast",
},
stages: 1,
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
total_usd_micros: 240000,
}),
usage: makeUsage({ input: 1200, output: 300 }, 240000),
},
],
}),
@ -249,7 +296,7 @@ describe("RunBilling", () => {
const footers = renderer.root.findAll((node) => node.type === "tfoot");
const footerCells = footers[0].findAll((node) => node.type === "td");
// The Run time column in the footer is index 3 (Total / [empty Model] /
// Tokens / Run time / Billing).
// Tokens / Run time / Cost).
const footerRuntime = textFromInstance(footerCells[3]);
expect(footerRuntime).toContain("30s");
} finally {

View file

@ -2,24 +2,25 @@ import { Fragment, useMemo } from "react";
import { EmptyState } from "../components/state";
import { Tooltip } from "../components/ui";
import {
billableOutputTokens,
billingTokenBuckets,
hasBillingUsage,
} from "../lib/billing";
import {
formatDurationMs,
formatTokenCount,
formatUsdMicros,
} from "../lib/format";
import { useRunBilling } from "../lib/queries";
import { useRunUsage } from "../lib/queries";
import { IN_FLIGHT_STAGE_STATES } from "../lib/stage-sidebar";
import { useTickingNow } from "../lib/time";
import {
billableOutputTokens,
costSourceTag,
hasUsage,
usageTokenBuckets,
} from "../lib/usage";
import type {
BilledTokenCounts,
BillingModelRef,
RunBilling,
RunBillingStage,
RunUsage,
RunUsageStage,
Usage,
UsageModelRef,
} from "@qltysh/fabro-api-client";
const EMPTY_VALUE = "—";
@ -29,34 +30,30 @@ function formatTokens(n: number | null | undefined) {
return formatTokenCount(n, { compactDecimal: true });
}
function formatUsdMicrosOrDash(usdMicros?: number | null): string {
return formatUsdMicros(usdMicros) ?? EMPTY_VALUE;
}
function formatModelRef(model?: BillingModelRef | null): string | null {
function formatModelRef(model?: UsageModelRef | null): string | null {
if (!model) return null;
const speed = model.speed ? ` · ${model.speed}` : "";
return `${model.provider}:${model.model_id}${speed}`;
}
function isInFlight(stage: RunBillingStage): boolean {
function isInFlight(stage: RunUsageStage): boolean {
return stage.state != null && IN_FLIGHT_STAGE_STATES.has(stage.state);
}
function isVisibleRow(row: MappedStageRow): boolean {
if (row.inFlight) return true;
return row.billing != null && hasBillingUsage(row.billing);
return row.usage != null && hasUsage(row.usage);
}
interface MappedStageRow {
stage: string;
model: string | null;
billing: BilledTokenCounts | null;
usage: Usage | null;
wallTimeMs: number;
inFlight: boolean;
}
function liveWallTimeMs(stage: RunBillingStage, now: number): number {
function liveWallTimeMs(stage: RunUsageStage, now: number): number {
if (stage.started_at) {
const startedMs = new Date(stage.started_at).getTime();
if (Number.isFinite(startedMs)) {
@ -68,20 +65,20 @@ function liveWallTimeMs(stage: RunBillingStage, now: number): number {
export const handle = { wide: true };
function mapStageRow(stage: RunBillingStage, wallTimeMs: number): MappedStageRow {
function mapStageRow(stage: RunUsageStage, wallTimeMs: number): MappedStageRow {
const hasModel = stage.model != null;
return {
stage: stage.stage.name,
model: formatModelRef(stage.model),
billing: hasModel ? stage.billing : null,
usage: hasModel ? stage.usage : null,
wallTimeMs,
inFlight: isInFlight(stage),
};
}
/** Hover breakdown of the disjoint token buckets behind an `in / out` count. */
function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
const buckets = billingTokenBuckets(billing);
function TokenBreakdown({ usage }: { usage: Usage }) {
const buckets = usageTokenBuckets(usage);
return (
<div className="min-w-44 py-0.5">
<div className="border-line text-fg-2 mb-1.5 border-b pb-1 font-medium">
@ -108,71 +105,94 @@ function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
* Renders an `input / output` token count. When the row has model usage,
* hovering the count reveals the cache breakdown.
*/
function TokensCell({ billing }: { billing: BilledTokenCounts | null }) {
function TokensCell({ usage }: { usage: Usage | null }) {
const display = (
<>
{formatTokens(billing?.input_tokens)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(billing ? billableOutputTokens(billing) : null)}
{formatTokens(usage?.tokens.input)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(usage ? billableOutputTokens(usage.tokens) : null)}
</>
);
if (!billing) return display;
if (!usage) return display;
return (
<Tooltip label={<TokenBreakdown billing={billing} />}>
<Tooltip label={<TokenBreakdown usage={usage} />}>
<span>{display}</span>
</Tooltip>
);
}
export default function RunBilling({ params }: { params: { id: string } }) {
const billingQuery = useRunBilling(params.id);
const billing = billingQuery.data;
const hasInFlight = billing?.stages.some(isInFlight) ?? false;
/**
* Renders a cost. A row with no model usage shows a dash; a usage whose cost
* is unknown (a model the catalog cannot price, or a total with an unpriced
* part) says so rather than showing zero. A cost that did not come from the
* catalog is tagged with where it came from.
*/
function CostCell({ usage }: { usage: Usage | null | undefined }) {
if (!usage) return <>{EMPTY_VALUE}</>;
const cost = usage.cost;
const formatted = formatUsdMicros(cost?.usd_micros);
if (formatted == null) return <span className="text-fg-muted">unknown</span>;
const tag = costSourceTag(cost);
return (
<>
{formatted}
{tag ? (
<span className="text-fg-muted ml-1.5 font-sans text-[10px] uppercase tracking-wide">
{tag}
</span>
) : null}
</>
);
}
export default function RunUsageRoute({ params }: { params: { id: string } }) {
const usageQuery = useRunUsage(params.id);
const runUsage: RunUsage | undefined = usageQuery.data;
const hasInFlight = runUsage?.stages.some(isInFlight) ?? false;
// Tick once per second only while a stage is in-flight.
const now = useTickingNow(hasInFlight);
// Completed rows don't depend on `now`; memoize them by `billing` so we
// Completed rows don't depend on `now`; memoize them by `runUsage` so we
// don't reallocate them every tick.
const completedRows = useMemo<MappedStageRow[]>(() => {
if (!billing) return [];
return billing.stages.map((stage) => mapStageRow(stage, stage.timing.wall_time_ms));
}, [billing]);
if (!runUsage) return [];
return runUsage.stages.map((stage) => mapStageRow(stage, stage.timing.wall_time_ms));
}, [runUsage]);
// The model breakdown is server-derived and stable across ticks too.
const modelBreakdown = useMemo(() => {
if (!billing) return [];
return billing.by_model
if (!runUsage) return [];
return runUsage.by_model
.map((entry) => ({
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
billing: entry.billing,
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
usage: entry.usage,
}))
.sort(
(a, b) =>
(b.billing.total_usd_micros ?? -1) - (a.billing.total_usd_micros ?? -1),
(b.usage.cost?.usd_micros ?? -1) - (a.usage.cost?.usd_micros ?? -1),
);
}, [billing]);
}, [runUsage]);
// Re-derive only the in-flight rows on each tick; everything else stays put.
const rows = useMemo<MappedStageRow[]>(() => {
if (!billing) return [];
if (!runUsage) return [];
if (!hasInFlight) return completedRows;
return billing.stages.map((stage, idx) =>
return runUsage.stages.map((stage, idx) =>
isInFlight(stage)
? mapStageRow(stage, liveWallTimeMs(stage, now))
: completedRows[idx],
);
}, [billing, completedRows, hasInFlight, now]);
}, [runUsage, completedRows, hasInFlight, now]);
// While ticking, sum the displayed row runtimes so the footer updates in
// lock-step. Otherwise trust the server's authoritative total.
const totalWallTimeMs = hasInFlight
? rows.reduce((sum, row) => sum + row.wallTimeMs, 0)
: (billing?.totals.timing.wall_time_ms ?? 0);
: (runUsage?.totals.timing.wall_time_ms ?? 0);
const hasLlmStages = (billing?.by_model.length ?? 0) > 0;
const totalBilling = hasLlmStages && billing ? billing.totals : null;
const totalUsdMicros = billing?.totals.total_usd_micros;
const hasLlmStages = (runUsage?.by_model.length ?? 0) > 0;
const totalUsage = hasLlmStages && runUsage ? runUsage.totals.usage : null;
const modelStageCount = modelBreakdown.reduce((sum, row) => sum + row.stages, 0);
const visibleRows = rows.filter(isVisibleRow);
@ -201,7 +221,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
<th className="px-4 py-2.5 font-medium">Model</th>
<th className="px-4 py-2.5 font-medium text-right">Tokens</th>
<th className="px-4 py-2.5 font-medium text-right">Run time</th>
<th className="px-4 py-2.5 font-medium text-right">Billing</th>
<th className="px-4 py-2.5 font-medium text-right">Cost</th>
</tr>
</thead>
<tbody>
@ -212,13 +232,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.model ?? EMPTY_VALUE}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell billing={row.billing} />
<TokensCell usage={row.usage} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatDurationMs(row.wallTimeMs)}
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.billing?.total_usd_micros)}
<CostCell usage={row.usage} />
</td>
</tr>
))}
@ -228,13 +248,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
<td className="px-4 py-3 font-medium text-fg">Total</td>
<td className="px-4 py-3 text-xs text-fg-muted">All models</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell billing={totalBilling} />
<TokensCell usage={totalUsage} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatDurationMs(totalWallTimeMs)}
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatUsdMicrosOrDash(totalUsdMicros)}
<CostCell usage={totalUsage} />
</td>
</tr>
</tfoot>
@ -251,7 +271,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
<th className="px-4 py-2.5 font-medium">Model</th>
<th className="px-4 py-2.5 font-medium text-right">Stages</th>
<th className="px-4 py-2.5 font-medium text-right">Tokens</th>
<th className="px-4 py-2.5 font-medium text-right">Billing</th>
<th className="px-4 py-2.5 font-medium text-right">Cost</th>
</tr>
</thead>
<tbody>
@ -262,10 +282,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.stages}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell billing={row.billing} />
<TokensCell usage={row.usage} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.billing.total_usd_micros)}
<CostCell usage={row.usage} />
</td>
</tr>
))}
@ -277,10 +297,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{modelStageCount}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell billing={totalBilling} />
<TokensCell usage={totalUsage} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatUsdMicrosOrDash(totalUsdMicros)}
<CostCell usage={totalUsage} />
</td>
</tr>
</tfoot>

View file

@ -5,7 +5,7 @@ import type { PaginatedRunList, Run } from "@qltysh/fabro-api-client";
import { ToastProvider } from "../components/toast";
import { CHILD_RUNS_LIST_PREFERENCES_STORAGE_KEY } from "../components/runs-list/preferences";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
import { setupReactTestEnv } from "../lib/test-utils";
class MemoryStorage {
@ -57,7 +57,7 @@ function run(id: string, repo = "qlty/fabro", workflow = "release"): Run {
last_event_at: "2026-04-19T12:04:00Z",
completed_at: "2026-04-19T12:05:00Z",
},
billing: null,
usage: makeUsage(),
size: "XS",
diff: null,
pull_request: null,

View file

@ -13,7 +13,7 @@ import {
} from "./runs";
import { summarizeBatchLifecycleAction } from "../components/runs-list/batch-lifecycle";
import { mapRunListItem } from "../data/runs";
import { TEST_PRINCIPAL } from "../lib/test-fixtures";
import { TEST_PRINCIPAL, makeUsage } from "../lib/test-fixtures";
function boardRun(id: string, column: BoardColumn, questionText?: string): Run {
const status =
@ -58,7 +58,7 @@ function boardRun(id: string, column: BoardColumn, questionText?: string): Run {
last_event_at: null,
completed_at: null,
},
billing: null,
usage: makeUsage(),
size: "XS",
diff: null,
pull_request: null,

View file

@ -138,16 +138,16 @@ Emitted when the workflow run finishes successfully (or with partial success).
"duration_ms": 45000,
"artifact_count": 3,
"status": "succeeded",
"total_cost": 0.15,
"final_git_commit_sha": "def456...",
"usage": {
"input_tokens": 15000,
"output_tokens": 5000,
"total_tokens": 20000,
"reasoning_tokens": 2000,
"cache_read_tokens": 8000,
"cache_write_tokens": 3000,
"speed": "standard"
"tokens": {
"input": 15000,
"output": 5000,
"reasoning": 2000,
"cache_read": 8000,
"cache_write": 3000
},
"cost": { "usd_micros": 150000, "source": "catalog" }
}
}
}
@ -158,17 +158,10 @@ Emitted when the workflow run finishes successfully (or with partial success).
| `duration_ms` | number | Total run duration in milliseconds |
| `artifact_count` | number | Number of artifacts produced |
| `status` | string | Final stage outcome (`"succeeded"`, `"failed"`, `"partially_succeeded"`, `"skipped"`) |
| `total_cost` | number? | Aggregate cost in USD |
| `final_git_commit_sha` | string? | Final HEAD SHA |
| `usage` | object? | Aggregate token usage |
| `usage.input_tokens` | number | Total input tokens |
| `usage.output_tokens` | number | Total output tokens |
| `usage.total_tokens` | number | Total tokens (input + output) |
| `usage.reasoning_tokens` | number? | Total reasoning/thinking tokens |
| `usage.cache_read_tokens` | number? | Total cache read tokens |
| `usage.cache_write_tokens` | number? | Total cache write tokens |
| `usage.speed` | string? | Speed tier |
| `usage.raw` | object? | Raw provider-specific usage data |
| `usage` | object? | The run's usage summed across every stage visit, as lithos-llm's `Usage`. Absent for a run that made no model calls |
| `usage.tokens` | object | The five disjoint token buckets: `input`, `output`, `reasoning`, `cache_read`, `cache_write`. Their plain sum is the total |
| `usage.cost` | object? | `usd_micros` and `source` (`catalog`, `provider`, or `application`). Absent when the cost is unknown, never zero: a sum has a cost only when every part that used tokens was priced |
### `run.failed`
@ -379,15 +372,19 @@ Emitted when a workflow node finishes execution.
"preferred_label": "tests_pass",
"suggested_next_ids": ["review"],
"usage": {
"model": "claude-sonnet-4-20250514",
"input_tokens": 5000,
"output_tokens": 2000,
"cache_read_tokens": 3000,
"cache_write_tokens": 1000,
"reasoning_tokens": 500,
"speed": "standard",
"cost": 0.05
"model": { "provider": "anthropic", "model_id": "claude-sonnet-4-20250514" },
"usage": {
"tokens": {
"input": 5000,
"output": 2000,
"reasoning": 500,
"cache_read": 3000,
"cache_write": 1000
},
"cost": { "usd_micros": 50000, "source": "catalog" }
}
},
"usage_by_model": [],
"error": "lint failed",
"failure_class": "deterministic",
"failure_signature": "clippy::unused_import",
@ -413,16 +410,10 @@ Emitted when a workflow node finishes execution.
| `status` | string | `"succeeded"`, `"failed"`, `"skipped"`, `"partially_succeeded"` |
| `preferred_label` | string? | Edge label hint for routing |
| `suggested_next_ids` | string[] | Suggested successor node ids |
| `usage` | object? | Token usage for this stage |
| `usage.model` | string | Model identifier |
| `usage.input_tokens` | number | Input tokens |
| `usage.output_tokens` | number | Output tokens |
| `usage.cache_read_tokens` | number? | Cache read tokens |
| `usage.cache_write_tokens` | number? | Cache write tokens |
| `usage.reasoning_tokens` | number? | Reasoning/thinking tokens |
| `usage.speed` | string? | Speed tier |
| `usage.cost` | number? | Estimated cost in USD |
| `billing_by_model` | array? | For an agent stage, the stage's billing split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know billed at the root's. Each row has `model`, `tokens`, and `total_usd_micros`, and the rows sum to the stage's billing. Empty for stages without a coding agent and on events written before it existed |
| `usage` | object? | The stage's usage under the model it ran on (`ModelUsage`): for an agent stage, the whole session tree's tokens under the root's route. Absent for a stage that made no model calls |
| `usage.model` | object | `provider`, `model_id`, and optional `speed` tier |
| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost is the provider's reported figure when it gave one, else the catalog's price; absent when the catalog has no rates |
| `usage_by_model` | array? | For an agent stage, `usage` split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know priced at the root's. Each row is a `ModelUsage`, and the rows sum to `usage`. Empty for stages without a coding agent and on events written before it existed |
| `error` | string? | Error message (flattened from failure detail) |
| `failure_class` | string? | `"transient_infra"`, `"deterministic"`, `"budget_exhausted"`, `"compilation_loop"`, `"canceled"`, `"structural"` |
| `failure_signature` | string? | Dedup key for repeated failures |
@ -473,8 +464,8 @@ Emitted when a stage fails (before retry decision).
| `failure_class` | string | Failure category |
| `failure_signature` | string? | Dedup key for repeated failures |
| `will_retry` | boolean | Whether the stage will be retried |
| `billing` | object? | What the stage spent before it failed, in the shape `stage.completed` uses. An agent stage that fails for good after answering model calls bills its whole session tree, as it would have on completion; a retried attempt and a cancelled stage carry none |
| `billing_by_model` | array? | `billing` split by model, as on `stage.completed` |
| `usage` | object? | What the stage spent before it failed, in the shape `stage.completed` uses. An agent stage that fails for good after answering model calls records its whole session tree, as it would have on completion; a retried attempt and a cancelled stage carry none |
| `usage_by_model` | array? | `usage` split by model, as on `stage.completed` |
### `stage.retrying`
@ -1085,12 +1076,14 @@ Emitted when the assistant produces a complete message.
"text": "I've fixed the bug in auth.rs by...",
"model": "claude-sonnet-4-20250514",
"usage": {
"input_tokens": 3000,
"output_tokens": 1500,
"total_tokens": 4500,
"reasoning_tokens": 200,
"cache_read_tokens": 1000,
"cache_write_tokens": 500
"tokens": {
"input": 3000,
"output": 1500,
"reasoning": 200,
"cache_read": 1000,
"cache_write": 500
},
"cost": { "usd_micros": 12500, "source": "provider" }
},
"tool_call_count": 2
}
@ -1101,13 +1094,9 @@ Emitted when the assistant produces a complete message.
|----------|------|-------------|
| `text` | string | Assistant message text |
| `model` | string | Model identifier |
| `usage` | object | Token usage for this message |
| `usage.input_tokens` | number | Input tokens |
| `usage.output_tokens` | number | Output tokens |
| `usage.total_tokens` | number | Total tokens |
| `usage.reasoning_tokens` | number? | Reasoning tokens |
| `usage.cache_read_tokens` | number? | Cache read tokens |
| `usage.cache_write_tokens` | number? | Cache write tokens |
| `usage` | object | lithos-llm's `Usage` for this message, as pebble reported it |
| `usage.tokens` | object | The five disjoint token buckets: `input`, `output`, `reasoning`, `cache_read`, `cache_write` |
| `usage.cost` | object? | `usd_micros` and `source`, when the provider reported a cost |
| `usage.speed` | string? | Speed tier |
| `usage.raw` | object? | Raw provider-specific usage |
| `tool_call_count` | number | Number of tool calls in this turn |

View file

@ -35,8 +35,8 @@ tags:
description: Workflow definitions and execution
- name: Workflow Versions
description: Immutable, content-addressed workflow packages
- name: Billing
description: Token counts and billed totals
- name: Usage
description: Token counts and costs
- name: Insights
description: SQL query editor and history
- name: Models
@ -3714,21 +3714,21 @@ paths:
schema:
$ref: "#/components/schemas/ErrorResponse"
/api/v1/runs/{id}/billing:
/api/v1/runs/{id}/usage:
get:
operationId: retrieveRunBilling
operationId: retrieveRunUsage
tags: [Run Outputs]
summary: Retrieve Run Billing
description: Returns token counts and billed totals broken down by stage and model for a specific run.
summary: Retrieve Run Usage
description: Returns token counts and costs broken down by stage and model for a specific run.
parameters:
- $ref: "#/components/parameters/RunId"
responses:
"200":
description: Billing data
description: Usage data
content:
application/json:
schema:
$ref: "#/components/schemas/RunBilling"
$ref: "#/components/schemas/RunUsage"
"404":
description: Run not found
headers:
@ -5224,21 +5224,21 @@ paths:
schema:
$ref: "#/components/schemas/PaginatedHistoryEntryList"
# ── Billing ──────────────────────────────────────────────────────────
# ── Usage ────────────────────────────────────────────────────────────
/api/v1/billing:
/api/v1/usage:
get:
operationId: getAggregateBilling
tags: [Billing]
summary: Aggregate Billing
description: Returns aggregate token counts and billed totals across all completed runs since server start.
operationId: getAggregateUsage
tags: [Usage]
summary: Aggregate Usage
description: Returns aggregate token counts and costs across all completed runs since server start.
responses:
"200":
description: Aggregate billing data
description: Aggregate usage data
content:
application/json:
schema:
$ref: "#/components/schemas/AggregateBilling"
$ref: "#/components/schemas/AggregateUsage"
# ── System ───────────────────────────────────────────────────────────
@ -8815,7 +8815,7 @@ components:
$ref: "#/components/schemas/ReasoningEffort"
description: Reasoning effort level.
speed:
$ref: "#/components/schemas/BillingSpeed"
$ref: "#/components/schemas/Speed"
description: Requested speed tier.
metadata:
type: object
@ -8827,38 +8827,45 @@ components:
description: Raw provider options keyed by provider id.
additionalProperties: true
CompletionUsage:
TokenCounts:
description: >
lithos `TokenCounts`: five disjoint token buckets for one completion.
`input` excludes cache reads and writes, while `output` excludes
reasoning tokens when the provider reports them separately.
lithos `TokenCounts`: five disjoint token buckets. Every token is
counted in exactly one, so their plain sum is the total. `input`
excludes cache reads and writes, while `output` excludes reasoning
tokens when the provider reports them separately. A bucket that is
absent reads as zero.
type: object
properties:
input:
type: integer
format: int64
format: uint64
minimum: 0
default: 0
description: Uncached prompt tokens.
description: Prompt tokens that were neither read from nor written to a cache.
output:
type: integer
format: int64
format: uint64
minimum: 0
default: 0
description: Non-reasoning completion tokens.
description: Completion tokens that are not reasoning tokens.
reasoning:
type: integer
format: int64
format: uint64
minimum: 0
default: 0
description: Separately reported reasoning tokens.
description: Completion tokens spent on reasoning, priced at the output rate.
cache_read:
type: integer
format: int64
format: uint64
minimum: 0
default: 0
description: Prompt tokens served from a provider cache.
cache_write:
type: integer
format: int64
format: uint64
minimum: 0
default: 0
description: Prompt tokens written to a provider cache.
description: Prompt tokens written into a provider cache.
ModelHandle:
description: A resolved provider and model identity.
@ -8871,22 +8878,48 @@ components:
type: string
description: Canonical model id within the provider.
CompletionCost:
Cost:
description: "lithos `Cost`: a USD amount in micros and where it came from."
type: object
required: [usd_micros, source]
properties:
usd_micros:
type: integer
format: int64
format: uint64
minimum: 0
source:
$ref: "#/components/schemas/CostSource"
Usage:
description: >-
lithos `Usage`: token counts and, when known, what they cost. `cost`
is absent when there is no cost data, never zero. A sum has a cost
only when every part that used tokens was priced; its `source` is the
parts' shared source, or `application` when they differ.
type: object
required: [tokens]
properties:
tokens:
$ref: "#/components/schemas/TokenCounts"
cost:
$ref: "#/components/schemas/Cost"
ModelUsage:
description: >-
Usage grouped under one model: one response, or one model's share of
a stage.
type: object
required: [model, usage]
properties:
model:
$ref: "#/components/schemas/UsageModelRef"
usage:
$ref: "#/components/schemas/Usage"
CompletionResponse:
description: >-
A lithos `Response`, returned verbatim. The server is the billing
authority: `cost` is the catalog estimate or the provider's own
A lithos `Response`, returned verbatim. The server prices the
response: `cost` is the catalog estimate or the provider's own
figure. When the request carried `schema`, `output` holds the parsed
object.
type: object
@ -8912,9 +8945,9 @@ components:
type: string
description: "Why generation stopped: stop, length, tool_call, content_filter, error, incomplete, or a provider-specific reason."
usage:
$ref: "#/components/schemas/CompletionUsage"
$ref: "#/components/schemas/TokenCounts"
cost:
$ref: "#/components/schemas/CompletionCost"
$ref: "#/components/schemas/Cost"
rate_limits:
type: object
additionalProperties: true
@ -8935,7 +8968,8 @@ components:
type: string
description: >
Where a cost came from: `catalog` (estimated from catalog prices),
`provider` (the provider's own billing data), or `application`.
`provider` (the provider's own reported cost), or `application`
(a sum the caller assembled from differently sourced parts).
enum: [catalog, provider, application]
PaginatedSavedQueryList:
@ -10424,7 +10458,7 @@ components:
- type: "null"
speed:
oneOf:
- $ref: "#/components/schemas/BillingSpeed"
- $ref: "#/components/schemas/Speed"
- type: "null"
permission_level:
oneOf:
@ -11159,10 +11193,15 @@ components:
Open tool batch: when the batch started and which calls have not
yet reported completion.
usage:
$ref: "#/components/schemas/BilledTokenCounts"
$ref: "#/components/schemas/Usage"
description: >-
The stage's usage: while the stage runs, its agent's own
accounting of the session tree with whatever cost the provider
reported; once it ends, the same tokens with the catalog's price
where the provider reported none.
model:
oneOf:
- $ref: "#/components/schemas/BillingModelRef"
- $ref: "#/components/schemas/UsageModelRef"
- type: "null"
permission_level:
oneOf:
@ -11191,18 +11230,18 @@ components:
Start of an external ACP agent process, if one is running. ACP
agents do not expose Fabro's internal LLM brackets, so the process
lifetime supplies their live inference estimate.
billing_by_model:
usage_by_model:
type: array
items:
$ref: "#/components/schemas/BilledModelUsage"
$ref: "#/components/schemas/ModelUsage"
default: []
description: >-
The completed stage's `usage` split by model, as `stage.completed`
reported it: the root session's route and each subagent's own
model, a subagent whose model the catalog does not know billed at
model, a subagent whose model the catalog does not know priced at
the root's. Sums to `usage`. Empty while the stage runs and for
stages without a coding agent; the billing rollup then bills
`usage` to `model`.
stages without a coding agent; the usage rollup then puts `usage`
under `model`.
agent:
oneOf:
- $ref: "#/components/schemas/AgentSessionProjection"
@ -11418,15 +11457,14 @@ components:
provider-reported cost for the root session and each descendant, the
route and where it moved, the context window, tools, MCP servers,
skills, todo lists, subagents, compactions, files touched, and the
prompt in progress. Counts only; pricing a count from the catalog is
fabro's, and lives in `StageProjection.usage`.
prompt in progress. Its costs are the provider's own; pricing from
the catalog is fabro's, and lives in `StageProjection.usage`.
type: object
required:
- root_session_id
- route
- activity
- usage
- cost_usd_micros
- messages
- descendants
- context_window
@ -11450,13 +11488,10 @@ components:
activity:
$ref: "#/components/schemas/AgentSessionActivity"
usage:
$ref: "#/components/schemas/TokenUsage"
description: The root session's usage over the stage.
cost_usd_micros:
type: ["integer", "null"]
format: uint64
minimum: 0
description: The root session's provider-reported cost, when a provider reported one.
$ref: "#/components/schemas/Usage"
description: >-
The root session's usage over the stage, with the provider's
reported cost when every answer carried one.
messages:
type: integer
format: uint64
@ -11573,7 +11608,6 @@ components:
required:
- parent
- usage
- cost_usd_micros
- messages
- compactions
properties:
@ -11589,11 +11623,7 @@ components:
The model it runs on, from its `SessionStarted`; when the start
was not seen, the model of its first answer.
usage:
$ref: "#/components/schemas/TokenUsage"
cost_usd_micros:
type: ["integer", "null"]
format: uint64
minimum: 0
$ref: "#/components/schemas/Usage"
messages:
type: integer
format: uint64
@ -11826,16 +11856,11 @@ components:
type: integer
minimum: 0
usage:
$ref: "#/components/schemas/TokenUsage"
$ref: "#/components/schemas/Usage"
description: >-
The summary call's tokens: a breakdown of the session's and the
prompt's usage, which already include them. Zero on compactions
The summary call's usage: a breakdown of the session's and the
prompt's usage, which already include it. Zero on compactions
recorded before it was kept.
cost_usd_micros:
type: integer
format: uint64
minimum: 0
description: The summary call's provider-reported cost, included in the totals the same way.
AgentSessionRouteFailover:
description: One move the root session made to a fallback route, as the stream reported it from the route it moved to.
@ -11865,15 +11890,11 @@ components:
$ref: "#/components/schemas/AgentErrorData"
description: The failure that ended the previous route.
usage:
$ref: "#/components/schemas/TokenUsage"
$ref: "#/components/schemas/Usage"
description: >-
What the prompt spent on the failed route. Already in the
session's and the prompt's totals through that route's committed
answers: a breakdown, not an addition.
cost_usd_micros:
type: integer
format: uint64
minimum: 0
inference_ms:
type: integer
format: uint64
@ -11918,7 +11939,6 @@ components:
required:
- completed
- usage
- cost_usd_micros
- messages
- context_window
- tool_calls
@ -11932,12 +11952,8 @@ components:
type: boolean
description: Whether the prompt reached its end.
usage:
$ref: "#/components/schemas/TokenUsage"
$ref: "#/components/schemas/Usage"
description: The root session's usage over the prompt.
cost_usd_micros:
type: ["integer", "null"]
format: uint64
minimum: 0
messages:
type: integer
format: uint64
@ -11986,44 +12002,6 @@ components:
last_file_touched:
type: ["string", "null"]
TokenUsage:
description: >-
Token accounting as the coding agent counts it. The five buckets are
disjoint: every token is counted in exactly one, so their plain sum
is the total. A bucket that is absent reads as zero.
type: object
properties:
input:
type: integer
format: uint64
minimum: 0
default: 0
description: Prompt tokens that were neither read from nor written to a cache.
output:
type: integer
format: uint64
minimum: 0
default: 0
description: Completion tokens that are not reasoning tokens.
reasoning:
type: integer
format: uint64
minimum: 0
default: 0
description: Completion tokens spent on reasoning, billed at the output rate.
cache_read:
type: integer
format: uint64
minimum: 0
default: 0
description: Prompt tokens served from a provider cache.
cache_write:
type: integer
format: uint64
minimum: 0
default: 0
description: Prompt tokens written into a provider cache.
McpToolSummary:
description: One tool an MCP server advertised, as the coding agent's registry named it.
type: object
@ -12207,7 +12185,7 @@ components:
- type: "null"
speed:
oneOf:
- $ref: "#/components/schemas/BillingSpeed"
- $ref: "#/components/schemas/Speed"
- type: "null"
InterviewOption:
@ -12411,6 +12389,7 @@ components:
- stage_id
- stage_label
- timing
- usage
- retries
properties:
stage_id:
@ -12419,9 +12398,9 @@ components:
type: string
timing:
$ref: "#/components/schemas/StageTiming"
billing_usd_micros:
type: ["integer", "null"]
format: int64
usage:
$ref: "#/components/schemas/Usage"
description: Per-node usage summed across every visit of the node.
retries:
type: integer
format: uint32
@ -12455,10 +12434,13 @@ components:
type: array
items:
$ref: "#/components/schemas/StageSummary"
billing:
usage:
oneOf:
- $ref: "#/components/schemas/BilledTokenCounts"
- $ref: "#/components/schemas/Usage"
- type: "null"
description: >-
The run's usage summed across every stage visit; null for a run
that made no model calls.
total_retries:
type: integer
format: uint32
@ -12654,7 +12636,7 @@ components:
- source_directory
- timestamps
- timing
- billing
- usage
- size
- ask_fabro
- diff
@ -12718,10 +12700,11 @@ components:
description: |
Run-level timing rollup. Wall time is the run's clock duration;
active timing sums work across stage visits.
billing:
oneOf:
- $ref: "#/components/schemas/RunBillingSummary"
- type: "null"
usage:
$ref: "#/components/schemas/Usage"
description: >-
The run's usage summed across every stage visit so far: the
conclusion's total once the run ended, else the sum of the stages'.
size:
$ref: "#/components/schemas/RunSize"
ask_fabro:
@ -12886,18 +12869,10 @@ components:
type: ["string", "null"]
format: date-time
RunBillingSummary:
type: object
required: [total_usd_micros]
properties:
total_usd_micros:
type: ["integer", "null"]
format: int64
RunSize:
type: string
enum: [XS, S, M, L, XL]
description: Run size bucket derived from current best-effort billed usage.
description: Run size bucket derived from the run's current cost.
RunLinks:
type: object
@ -13089,75 +13064,10 @@ components:
type: string
enum: [github, git, unknown]
BilledTokenCounts:
description: Token counts with optional billed USD micros totals.
type: object
required:
- input_tokens
- output_tokens
- total_tokens
- reasoning_tokens
- cache_read_tokens
- cache_write_tokens
properties:
input_tokens:
type: integer
format: int64
description: Number of input tokens consumed.
example: 28640
output_tokens:
type: integer
format: int64
description: Number of output tokens generated.
example: 8750
total_tokens:
type: integer
format: int64
description: Total billable tokens aggregated across categories.
example: 37390
reasoning_tokens:
type: integer
format: int64
description: Number of reasoning tokens.
example: 1200
cache_read_tokens:
type: integer
format: int64
description: Number of cache read tokens.
example: 4800
cache_write_tokens:
type: integer
format: int64
description: Number of cache write tokens.
example: 1500
total_usd_micros:
type: ["integer", "null"]
format: int64
description: Billed USD amount in micros.
example: 720000
BilledModelUsage:
UsageModelRef:
description: >-
Usage and cost billed to one model: one response, or one model's share
of a stage.
type: object
required:
- model
- tokens
properties:
model:
$ref: "#/components/schemas/BillingModelRef"
tokens:
$ref: "#/components/schemas/CompletionUsage"
total_usd_micros:
type: integer
format: int64
description: >-
Cost for `tokens`, when the provider reported one or the catalog
could price them. Absent means no cost data, not zero.
BillingModelRef:
description: Provider-qualified billing model identity used for cost estimates.
Provider-qualified model identity a usage is grouped under. Carries
the requested speed tier because providers price tiers differently.
type: object
required:
- provider
@ -13169,10 +13079,10 @@ components:
type: string
speed:
oneOf:
- $ref: "#/components/schemas/BillingSpeed"
- $ref: "#/components/schemas/Speed"
- type: "null"
BillingSpeed:
Speed:
description: "lithos `Speed`: the requested latency or cost tier."
type: string
enum:
@ -13709,52 +13619,21 @@ components:
description: Question text.
example: Accept or push for another round?
AggregateBillingTotals:
description: Aggregate billing totals across all runs.
AggregateUsageTotals:
description: Aggregate usage totals across all runs.
type: object
required:
- runs
- input_tokens
- output_tokens
- total_tokens
- reasoning_tokens
- cache_read_tokens
- cache_write_tokens
- usage
- timing
properties:
runs:
type: integer
description: Total number of completed runs.
example: 9
input_tokens:
type: integer
description: Total input tokens.
example: 643860
output_tokens:
type: integer
description: Total output tokens.
example: 189720
total_tokens:
type: integer
description: Total tokens aggregated across all billing categories.
example: 833580
reasoning_tokens:
type: integer
description: Total reasoning tokens.
example: 12040
cache_read_tokens:
type: integer
description: Total cache read tokens.
example: 85400
cache_write_tokens:
type: integer
description: Total cache write tokens.
example: 9200
total_usd_micros:
type: ["integer", "null"]
format: int64
description: Total billed USD amount in micros.
example: 20340000
usage:
$ref: "#/components/schemas/Usage"
description: Tokens and cost summed across every completed run.
timing:
$ref: "#/components/schemas/RunTiming"
description: |
@ -13762,8 +13641,8 @@ components:
sums work across stage visits, so `active_time_ms` can exceed
`wall_time_ms`.
BillingStageRef:
description: Reference to a workflow node in a billing stage row.
UsageStageRef:
description: Reference to a workflow node in a usage stage row.
type: object
required:
- id
@ -13881,7 +13760,7 @@ components:
- status
- node_id
- visit
- billing
- usage
properties:
id:
$ref: "#/components/schemas/StageId"
@ -13957,15 +13836,15 @@ components:
format: date-time
description: Wall-clock time the latest attempt of this stage started, if known.
example: "2026-04-29T12:34:56Z"
billing:
$ref: "#/components/schemas/BilledTokenCounts"
usage:
$ref: "#/components/schemas/Usage"
description: >-
Token counts for this stage execution alone. `total_usd_micros` is
the provider-reported cost when there is one, otherwise the server
catalog's price for these tokens — the same pricing the
`/runs/{id}/billing` rows use. All-zero counts mean the stage made
no model calls. Unlike the billing rows, which sum every visit of a
node, this covers only this visit.
Usage for this stage execution alone. `cost` is the provider's
reported cost when there is one, otherwise the server catalog's
price for these tokens — the same pricing the `/runs/{id}/usage`
rows use. All-zero counts mean the stage made no model calls.
Unlike the usage rows, which sum every visit of a node, this
covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────
@ -14300,26 +14179,26 @@ components:
meta:
$ref: "#/components/schemas/RunCommitsMeta"
# ── Billing Schemas ──────────────────────────────────────────────────
# ── Usage Schemas ────────────────────────────────────────────────────
RunBillingStage:
description: Token counts and billed totals for one workflow node within a run. Rows are grouped by node; billing and timing sum every visit of that node.
RunUsageStage:
description: Token counts and cost for one workflow node within a run. Rows are grouped by node; usage and timing sum every visit of that node.
type: object
required:
- stage
- model
- billing
- usage
- timing
properties:
stage:
$ref: "#/components/schemas/BillingStageRef"
$ref: "#/components/schemas/UsageStageRef"
model:
description: Latest usage-bearing visit model for this node; null when no visit used an LLM model.
oneOf:
- $ref: "#/components/schemas/BillingModelRef"
- $ref: "#/components/schemas/UsageModelRef"
- type: "null"
billing:
$ref: "#/components/schemas/BilledTokenCounts"
usage:
$ref: "#/components/schemas/Usage"
timing:
$ref: "#/components/schemas/StageTiming"
description: |
@ -14336,72 +14215,43 @@ components:
- type: "null"
description: Lifecycle state of the stage. Use to detect in-flight rows for client-side runtime ticking.
RunBillingTotals:
description: Aggregate billing totals across all stages of a run.
RunUsageTotals:
description: Aggregate usage totals across all stages of a run.
type: object
required:
- timing
- input_tokens
- output_tokens
- total_tokens
- reasoning_tokens
- cache_read_tokens
- cache_write_tokens
- usage
properties:
timing:
$ref: "#/components/schemas/RunTiming"
description: |
Run-level timing rollup. `wall_time_ms` is summed across stage
visits; active timing sums work across visits.
input_tokens:
type: integer
description: Total input tokens consumed.
example: 71540
output_tokens:
type: integer
description: Total output tokens generated.
example: 21080
total_tokens:
type: integer
description: Total tokens aggregated across all billing categories.
example: 92620
reasoning_tokens:
type: integer
description: Total reasoning tokens.
example: 3400
cache_read_tokens:
type: integer
description: Total cache read tokens.
example: 22000
cache_write_tokens:
type: integer
description: Total cache write tokens.
example: 4500
total_usd_micros:
type: ["integer", "null"]
format: int64
description: Total billed USD amount in micros.
example: 2260000
usage:
$ref: "#/components/schemas/Usage"
description: >-
Tokens and cost summed across every stage visit. The cost is
known only when every visit that used tokens was priced.
BillingByModel:
description: Billing statistics grouped by model.
UsageByModel:
description: Usage grouped by model.
type: object
required:
- model
- stages
- billing
- usage
properties:
model:
$ref: "#/components/schemas/BillingModelRef"
$ref: "#/components/schemas/UsageModelRef"
stages:
type: integer
description: Number of usage-bearing stage visits that used this model.
example: 2
billing:
$ref: "#/components/schemas/BilledTokenCounts"
usage:
$ref: "#/components/schemas/Usage"
RunBilling:
description: Complete billing breakdown for a single run.
RunUsage:
description: Complete usage breakdown for a single run.
type: object
required:
- stages
@ -14410,31 +14260,31 @@ components:
properties:
stages:
type: array
description: Per-node billing breakdown. Each row sums billing and runtime across all visits of that node.
description: Per-node usage breakdown. Each row sums usage and runtime across all visits of that node.
items:
$ref: "#/components/schemas/RunBillingStage"
$ref: "#/components/schemas/RunUsageStage"
totals:
$ref: "#/components/schemas/RunBillingTotals"
$ref: "#/components/schemas/RunUsageTotals"
by_model:
type: array
description: Billing grouped by model.
description: Usage grouped by model.
items:
$ref: "#/components/schemas/BillingByModel"
$ref: "#/components/schemas/UsageByModel"
AggregateBilling:
description: Aggregate token counts and billed totals across all runs since server start.
AggregateUsage:
description: Aggregate token counts and costs across all runs since server start.
type: object
required:
- totals
- by_model
properties:
totals:
$ref: "#/components/schemas/AggregateBillingTotals"
$ref: "#/components/schemas/AggregateUsageTotals"
by_model:
type: array
description: Billing grouped by model.
description: Usage grouped by model.
items:
$ref: "#/components/schemas/BillingByModel"
$ref: "#/components/schemas/UsageByModel"
PreviewUrlRequest:
description: Request body for generating a preview URL from a sandbox port.

View file

@ -230,7 +230,7 @@
"pages": [
"GET /api/v1/runs/{id}/artifacts",
"GET /api/v1/runs/{id}/artifacts/download",
"GET /api/v1/runs/{id}/billing",
"GET /api/v1/runs/{id}/usage",
{
"group": "Run Internals",
"icon": "microchip",

View file

@ -390,8 +390,11 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O
"succeeded" | "partially_succeeded" => &styles.bold_green,
_ => &styles.bold_red,
};
let usage = prop_field(envelope, "usage");
let cost = format_cost(
prop_field(envelope, "total_usd_micros")
usage
.and_then(|value| value.get("cost"))
.and_then(|value| value.get("usd_micros"))
.or_else(|| prop_field(envelope, "total_cost")),
);
@ -407,13 +410,12 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O
let mut lines = vec![summary];
if let Some(billing) =
prop_field(envelope, "billing").or_else(|| prop_field(envelope, "usage"))
{
let total = billing
.get("total_tokens")
.and_then(serde_json::Value::as_u64)
.unwrap_or(0);
if let Some(tokens) = usage.and_then(|value| value.get("tokens")) {
let bucket = |name: &str| tokens.get(name).and_then(serde_json::Value::as_u64);
let total = ["input", "output", "reasoning", "cache_read", "cache_write"]
.into_iter()
.filter_map(bucket)
.fold(0_u64, u64::saturating_add);
let pad = " ".repeat(ts.len() + 1);
if total > 0 {
lines.push(format!(
@ -424,14 +426,8 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O
.apply_to(format!("Tokens: {}", format_tokens(total)))
));
}
if let Some(cache_read) = billing
.get("cache_read_tokens")
.and_then(serde_json::Value::as_u64)
{
let cache_write = billing
.get("cache_write_tokens")
.and_then(serde_json::Value::as_u64)
.unwrap_or(0);
if let Some(cache_read) = bucket("cache_read") {
let cache_write = bucket("cache_write").unwrap_or(0);
lines.push(format!(
"{}{}",
pad,
@ -442,10 +438,7 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O
))
));
}
if let Some(reasoning) = billing
.get("reasoning_tokens")
.and_then(serde_json::Value::as_u64)
{
if let Some(reasoning) = bucket("reasoning") {
if reasoning > 0 {
lines.push(format!(
"{}{}",
@ -535,18 +528,20 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O
"stage.completed" => {
let label = str_field(envelope, "node_label").unwrap_or("?");
let duration = format_duration_ms(timing_wall_field(envelope));
let billing = prop_field(envelope, "billing").or_else(|| prop_field(envelope, "usage"));
// `stage.completed.usage` is a `ModelUsage`: the model, then the usage.
let usage = prop_field(envelope, "usage").and_then(|value| value.get("usage"));
let cost = format_cost(
billing
.and_then(|value| value.get("total_usd_micros"))
.or_else(|| billing.and_then(|value| value.get("cost"))),
usage
.and_then(|value| value.get("cost"))
.and_then(|value| value.get("usd_micros")),
);
let input_tokens = billing
.and_then(|value| value.get("input_tokens"))
let tokens = usage.and_then(|value| value.get("tokens"));
let input_tokens = tokens
.and_then(|value| value.get("input"))
.and_then(serde_json::Value::as_u64)
.unwrap_or(0);
let output_tokens = billing
.and_then(|value| value.get("output_tokens"))
let output_tokens = tokens
.and_then(|value| value.get("output"))
.and_then(serde_json::Value::as_u64)
.unwrap_or(0);
let token_total = input_tokens.saturating_add(output_tokens);
@ -921,7 +916,7 @@ fn format_duration_ms(value: Option<&serde_json::Value>) -> String {
fn format_cost(value: Option<&serde_json::Value>) -> String {
match value {
Some(value) => {
if let Some(usd_micros) = value.as_i64() {
if let Some(usd_micros) = value.as_u64() {
if usd_micros > 0 {
return format_usd_micros(usd_micros);
}
@ -1091,7 +1086,7 @@ mod tests {
#[test]
fn pretty_stage_completed() {
let styles = no_color_styles();
let line = r#"{"ts":"2026-01-01T14:23:15Z","event":"stage.completed","node_label":"plan","properties":{"timing":{"wall_time_ms":8000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"cost":0.12,"input_tokens":10000,"output_tokens":5200}}}"#;
let line = r#"{"ts":"2026-01-01T14:23:15Z","event":"stage.completed","node_label":"plan","properties":{"timing":{"wall_time_ms":8000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"model":{"provider":"openai","model_id":"gpt-5.4"},"usage":{"tokens":{"input":10000,"output":5200},"cost":{"usd_micros":120000,"source":"catalog"}}}}}"#;
let result = format_event_pretty(line, &styles).unwrap();
assert!(result.contains("plan"), "got: {result}");
assert!(result.contains("$0.12"), "got: {result}");
@ -1170,12 +1165,12 @@ mod tests {
#[test]
fn pretty_workflow_run_completed() {
let styles = no_color_styles();
let line = r#"{"ts":"2026-01-01T14:23:32Z","run_id":"abc123","event":"run.completed","properties":{"timing":{"wall_time_ms":25000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","total_usd_micros":570000,"billing":{"input_tokens":5000,"output_tokens":2000,"total_tokens":7000,"cache_read_tokens":3000,"cache_write_tokens":500,"reasoning_tokens":800}}}"#;
let line = r#"{"ts":"2026-01-01T14:23:32Z","run_id":"abc123","event":"run.completed","properties":{"timing":{"wall_time_ms":25000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"tokens":{"input":5000,"output":2000,"cache_read":3000,"cache_write":500,"reasoning":800},"cost":{"usd_micros":570000,"source":"catalog"}}}}"#;
let result = format_event_pretty(line, &styles).unwrap();
assert!(result.contains("SUCCEEDED"), "got: {result}");
assert!(result.contains("25s"), "got: {result}");
assert!(result.contains("$0.57"), "got: {result}");
assert!(result.contains("7.0k toks"), "got: {result}");
assert!(result.contains("11.3k toks"), "got: {result}");
assert!(result.contains("Cache:"), "got: {result}");
assert!(result.contains("3.0k toks read"), "got: {result}");
assert!(result.contains("Reasoning:"), "got: {result}");

View file

@ -208,10 +208,11 @@ pub(crate) fn print_run_conclusion(
HumanDuration(Duration::from_millis(conclusion.timing.wall_time_ms))
);
if let Some(billing) = conclusion.billing.as_ref() {
let total_tokens = billing.total_tokens;
if let Some(usage) = conclusion.usage {
let total_tokens = usage.total_tokens();
let cost_usd_micros = usage.cost.map(|cost| cost.usd_micros);
if total_tokens > 0 {
if let Some(total_usd_micros) = billing.total_usd_micros {
if let Some(total_usd_micros) = cost_usd_micros {
if total_usd_micros > 0 {
fabro_util::printerr!(
printer,
@ -232,28 +233,28 @@ pub(crate) fn print_run_conclusion(
.apply_to(format!("Toks: {}", format_tokens_human(total_tokens)))
);
}
if billing.cache_read_tokens > 0 || billing.cache_write_tokens > 0 {
if usage.tokens.cache_read > 0 || usage.tokens.cache_write > 0 {
fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(format!(
"Cache: {} read, {} write",
format_tokens_human(billing.cache_read_tokens),
format_tokens_human(billing.cache_write_tokens),
format_tokens_human(usage.tokens.cache_read),
format_tokens_human(usage.tokens.cache_write),
)),
);
}
if billing.reasoning_tokens > 0 {
if usage.tokens.reasoning > 0 {
fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(format!(
"Reasoning: {} tokens",
format_tokens_human(billing.reasoning_tokens),
format_tokens_human(usage.tokens.reasoning),
)),
);
}
} else if billing.total_usd_micros.is_none() {
} else if cost_usd_micros.is_none() {
fabro_util::printerr!(
printer,
"{}",

View file

@ -1,5 +1,5 @@
use chrono::{DateTime, Utc};
use fabro_types::{BilledModelUsage, EventBody, RunEvent};
use fabro_types::{EventBody, ModelUsage, RunEvent};
use fabro_util::{error, text};
use fabro_workflow::event::RunNoticeLevel;
use pebble_coding_agent::events::{CodingEvent, ErrorKind as AgentErrorKind, LlmOutputKind};
@ -13,12 +13,15 @@ pub(super) struct ProgressUsage {
}
impl ProgressUsage {
pub(super) fn from_stage_usage(usage: &BilledModelUsage) -> Self {
let tokens = usage.tokens();
pub(super) fn from_stage_usage(usage: &ModelUsage) -> Self {
let tokens = usage.usage.tokens;
Self {
input_tokens: tokens.input,
output_tokens: tokens.billable_output(),
cost: usage.total_usd_micros.map(|cost| cost as f64 / 1_000_000.0),
cost: usage
.usage
.cost
.map(|cost| cost.usd_micros as f64 / 1_000_000.0),
}
}
@ -294,7 +297,7 @@ pub(super) fn from_run_event(stored: &RunEvent) -> Option<ProgressEvent> {
name: node_label,
timing: props.timing,
status: props.status.to_string(),
usage: props.billing.as_ref().map(ProgressUsage::from_stage_usage),
usage: props.usage.as_ref().map(ProgressUsage::from_stage_usage),
}),
EventBody::StageFailed(props) => Some(ProgressEvent::StageFailed {
node_id,
@ -651,8 +654,8 @@ mod tests {
status: "succeeded".into(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),

View file

@ -461,12 +461,12 @@ mod tests {
use fabro_workflow::event::{
Event, RunNoticeLevel, SandboxLifecycle, to_run_event, to_run_event_at,
};
use fabro_workflow::outcome::billed_model_usage_from_llm;
use fabro_workflow::outcome::model_usage_from_llm;
use lithos_llm::catalog::{ModelId, builtin};
use lithos_llm::types::TokenCounts;
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, CompactionReason, ErrorData as AgentErrorData,
ErrorKind as AgentErrorKind, TokenUsage,
ErrorKind as AgentErrorKind, Usage,
};
use super::*;
@ -618,9 +618,7 @@ mod tests {
CodingEvent::AssistantMessage {
text: text.into(),
model: model.into(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -650,9 +648,9 @@ mod tests {
status: "succeeded".into(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: Some(
billed_model_usage_from_llm(
usage_by_model: Vec::new(),
usage: Some(
model_usage_from_llm(
&fabro_llm::test_support::test_catalog(),
&ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")),
TokenCounts {
@ -781,8 +779,7 @@ mod tests {
summary_token_estimate: 500,
tracked_file_count: 3,
reason: CompactionReason::Threshold,
usage: TokenUsage::default(),
cost_usd_micros: None,
usage: Usage::default(),
}),
);
assert!(ui.stage.active_stages["s1"].compaction_bar.is_none());

View file

@ -161,7 +161,6 @@ impl StageDisplay {
self.stage_counts.get(node_id).copied().unwrap_or((0, 0));
let total_tokens = usage.map_or(0, ProgressUsage::total_tokens);
if turn_count > 0 || tool_call_count > 0 || total_tokens > 0 {
let total_tokens = i64::try_from(total_tokens).unwrap_or(i64::MAX);
format!(
" {}",
renderer.styles().dim.apply_to(format!(

View file

@ -1341,11 +1341,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})),
Some(WorkerTitlePhase::Succeeded)
);
@ -1359,7 +1358,7 @@ mod tests {
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})),
Some(WorkerTitlePhase::Cancelled)
);
@ -1373,7 +1372,7 @@ mod tests {
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})),
Some(WorkerTitlePhase::Failed)
);

View file

@ -84,12 +84,9 @@ fn build_json_output(
if let Some(c) = conclusion {
value["timing"] =
serde_json::to_value(c.timing).unwrap_or_else(|_| serde_json::Value::Null);
if let Some(total_usd_micros) = c
.billing
.as_ref()
.and_then(|billing| billing.total_usd_micros)
{
value["total_usd_micros"] = total_usd_micros.into();
if let Some(usage) = c.usage {
value["usage"] =
serde_json::to_value(usage).unwrap_or_else(|_| serde_json::Value::Null);
}
}
value
@ -117,10 +114,9 @@ fn print_human_output(
Some(c) => {
let duration = format_duration_ms(c.timing.wall_time_ms);
let cost = c
.billing
.as_ref()
.and_then(|billing| billing.total_usd_micros)
.map(|value| format!(" {}", format_usd_micros(value)))
.usage
.and_then(|usage| usage.cost)
.map(|cost| format!(" {}", format_usd_micros(cost.usd_micros)))
.unwrap_or_default();
format!(" {duration}{cost}")
}
@ -138,13 +134,25 @@ fn print_human_output(
#[cfg(test)]
mod tests {
use fabro_types::{
BilledTokenCounts, FailureCategory, FailureDetail, FailureReason, RunDiff, RunFailure,
RunStatus, StageOutcome, SuccessReason, fixtures,
FailureCategory, FailureDetail, FailureReason, RunDiff, RunFailure, RunStatus,
StageOutcome, SuccessReason, fixtures,
};
use fabro_workflow::records::Conclusion;
use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage};
use super::*;
/// A usage with only a catalog cost.
fn priced(usd_micros: u64) -> Usage {
Usage {
tokens: TokenCounts::default(),
cost: Some(Cost {
usd_micros,
source: CostSource::Catalog,
}),
}
}
fn no_color_styles() -> Styles {
Styles::new(false)
}
@ -159,15 +167,7 @@ mod tests {
failure: None,
final_git_commit_sha: None,
stages: vec![],
billing: Some(BilledTokenCounts {
input_tokens: 0,
output_tokens: 0,
total_tokens: 0,
reasoning_tokens: 0,
cache_read_tokens: 0,
cache_write_tokens: 0,
total_usd_micros: Some(420_000),
}),
usage: Some(priced(420_000)),
total_retries: 0,
diff: RunDiff::default(),
};
@ -181,7 +181,8 @@ mod tests {
assert_eq!(json["run_id"], run_id.to_string());
assert_eq!(json["status"], "succeeded");
assert_eq!(json["timing"]["wall_time_ms"], 12345);
assert_eq!(json["total_usd_micros"], 420_000);
assert_eq!(json["usage"]["cost"]["usd_micros"], 420_000);
assert_eq!(json["usage"]["cost"]["source"], "catalog");
}
#[test]
@ -197,7 +198,7 @@ mod tests {
assert_eq!(json["run_id"], run_id.to_string());
assert_eq!(json["status"], "failed");
assert!(json.get("timing").is_none());
assert!(json.get("total_usd_micros").is_none());
assert!(json.get("usage").is_none());
}
#[test]
@ -221,7 +222,7 @@ mod tests {
}),
final_git_commit_sha: None,
stages: vec![],
billing: None,
usage: None,
total_retries: 0,
diff: RunDiff::default(),
};
@ -232,7 +233,7 @@ mod tests {
&run_id,
Some(&conclusion),
);
assert!(json.get("total_usd_micros").is_none());
assert!(json.get("usage").is_none());
assert_eq!(json["timing"]["wall_time_ms"], 500);
}
@ -247,15 +248,7 @@ mod tests {
failure: None,
final_git_commit_sha: None,
stages: vec![],
billing: Some(BilledTokenCounts {
input_tokens: 0,
output_tokens: 0,
total_tokens: 0,
reasoning_tokens: 0,
cache_read_tokens: 0,
cache_write_tokens: 0,
total_usd_micros: Some(150_000),
}),
usage: Some(priced(150_000)),
total_retries: 0,
diff: RunDiff::default(),
};

View file

@ -362,7 +362,7 @@ async fn main_inner(worker_token: Option<String>) -> (String, Result<()>) {
Box::pin(commands::pr::dispatch(ns, &base_ctx)).await?;
}
Commands::Parent(ns) => {
commands::parent::dispatch(ns, &base_ctx).await?;
Box::pin(commands::parent::dispatch(ns, &base_ctx)).await?;
}
Commands::Secret(ns) => {
commands::secret::dispatch(ns, &base_ctx).await?;

View file

@ -77,11 +77,8 @@ impl ServerRunInfo {
self.run.timing.as_ref().map(|t| t.wall_time_ms)
}
pub(crate) fn total_usd_micros(&self) -> Option<i64> {
self.run
.billing
.as_ref()
.and_then(|billing| billing.total_usd_micros)
pub(crate) fn total_usd_micros(&self) -> Option<u64> {
self.run.usage.cost.map(|cost| cost.usd_micros)
}
pub(crate) fn source_directory(&self) -> Option<&str> {

View file

@ -130,7 +130,7 @@ pub(crate) fn relative_path(path: &Path) -> String {
tilde_path(path)
}
pub(crate) fn format_tokens_human(tokens: i64) -> String {
pub(crate) fn format_tokens_human(tokens: u64) -> String {
if tokens >= 1_000_000 {
format!("{:.1}m", tokens as f64 / 1_000_000.0)
} else if tokens >= 1000 {
@ -140,7 +140,7 @@ pub(crate) fn format_tokens_human(tokens: i64) -> String {
}
}
pub(crate) fn format_usd_micros(usd_micros: i64) -> String {
pub(crate) fn format_usd_micros(usd_micros: u64) -> String {
format!("${:.2}", usd_micros as f64 / 1_000_000.0)
}

View file

@ -63,7 +63,7 @@ fn remote_run_state_response(run_id: &str) -> serde_json::Value {
"status": "succeeded",
"timing": {"wall_time_ms": 12, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0},
"stages": [],
"billing": null,
"usage": null,
"total_retries": 0,
"diff": {}
});

View file

@ -338,7 +338,7 @@ pub(crate) fn remote_run_summary_json(
"completed_at": null
},
"timing": null,
"billing": null,
"usage": {"tokens": {"input": 0, "output": 0, "reasoning": 0, "cache_read": 0, "cache_write": 0}},
"diff": null,
"pull_request": null,
"current_question": null,
@ -1253,10 +1253,9 @@ async fn append_seeded_simple_completion_events(
"artifact_count": 0,
"status": "succeeded",
"reason": "completed",
"total_usd_micros": null,
"final_git_commit_sha": null,
"final_patch": null,
"billing": null,
"usage": null,
}),
)
.await;
@ -1428,10 +1427,9 @@ async fn append_seeded_git_completion_events(
"artifact_count": 0,
"status": "succeeded",
"reason": "completed",
"total_usd_micros": null,
"final_git_commit_sha": step_two_sha,
"final_patch": final_story_patch(),
"billing": null,
"usage": null,
}),
)
.await;
@ -1498,10 +1496,9 @@ async fn append_seeded_git_noop_events(
"artifact_count": 0,
"status": "succeeded",
"reason": "completed",
"total_usd_micros": null,
"final_git_commit_sha": base_sha,
"final_patch": null,
"billing": null,
"usage": null,
}),
)
.await;
@ -1567,10 +1564,9 @@ async fn append_seeded_artifact_run_events(
"artifact_count": 7,
"status": "succeeded",
"reason": "completed",
"total_usd_micros": null,
"final_git_commit_sha": null,
"final_patch": null,
"billing": null,
"usage": null,
}),
)
.await;
@ -1728,7 +1724,7 @@ fn stage_completed_properties(index: usize, response: Option<&str>) -> serde_jso
"status": "succeeded",
"preferred_label": null,
"suggested_next_ids": [],
"billing": null,
"usage": null,
"failure": null,
"notes": null,
"files_touched": [],

View file

@ -335,12 +335,12 @@ fn demo_run_files() -> PaginatedRunFileList {
}
}
pub(crate) async fn get_run_billing(
pub(crate) async fn get_run_usage(
_auth: RequiredUser,
State(_state): State<Arc<AppState>>,
Path(_id): Path<String>,
) -> Response {
(StatusCode::OK, Json(runs::billing())).into_response()
(StatusCode::OK, Json(runs::usage())).into_response()
}
pub(crate) async fn get_run_settings(
@ -1073,11 +1073,11 @@ pub(crate) async fn prune_runs(
// ── Usage ──────────────────────────────────────────────────────────────
pub(crate) async fn get_aggregate_billing(
pub(crate) async fn get_aggregate_usage(
_auth: RequiredUser,
State(_state): State<Arc<AppState>>,
) -> Response {
(StatusCode::OK, Json(billing::aggregate())).into_response()
(StatusCode::OK, Json(usage::aggregate())).into_response()
}
// ── Data modules ───────────────────────────────────────────────────────
@ -1101,11 +1101,11 @@ mod runs {
};
use fabro_types::settings::{InterpString, ProjectNamespace, WorkflowNamespace};
use fabro_types::{
AuthMethod, IdpIdentity, PendingReason, Principal, RepositoryRef, RunBillingSummary, RunId,
RunLifecycle, RunLinks, RunOrigin, RunSize, RunTimestamps, StageId, WorkflowRef,
WorkflowSettings,
AuthMethod, IdpIdentity, PendingReason, Principal, RepositoryRef, RunId, RunLifecycle,
RunLinks, RunOrigin, RunSize, RunTimestamps, StageId, WorkflowRef, WorkflowSettings,
};
use lithos_llm::catalog::ProviderId;
use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage};
use super::ts;
@ -1124,14 +1124,29 @@ mod runs {
.collect()
}
fn billing_model(provider: ProviderId, model_id: &str) -> BillingModelRef {
BillingModelRef {
fn usage_model(provider: ProviderId, model_id: &str) -> UsageModelRef {
UsageModelRef {
provider,
model_id: model_id.into(),
speed: None,
}
}
/// Demo usage priced from the catalog.
fn priced(input: u64, output: u64, usd_micros: u64) -> Usage {
Usage {
tokens: TokenCounts {
input,
output,
..TokenCounts::default()
},
cost: Some(Cost {
usd_micros,
source: CostSource::Catalog,
}),
}
}
fn stage(
stage_id: &StageId,
name: &str,
@ -1143,7 +1158,7 @@ mod runs {
id: stage_id.clone(),
name: name.to_owned(),
handler,
billing: BilledTokenCounts::default(),
usage: Usage::default(),
status,
wall_time_ms,
node_id: stage_id.node_id().to_owned(),
@ -1188,7 +1203,7 @@ mod runs {
elapsed_secs: Option<f64>,
status_reason: Option<&str>,
pending_control: Option<RunControlAction>,
total_usd_micros: Option<i64>,
cost_usd_micros: Option<u64>,
entries: &[(&str, &str)],
) -> Run {
let created_at = ts(created_at);
@ -1197,6 +1212,10 @@ mod runs {
let repo_origin_url = Some(format!("https://github.com/demo/{repo_name}.git"));
let wall_time_ms = elapsed_secs.and_then(duration_ms_from_secs);
let timing = wall_time_ms.map(fabro_types::RunTiming::wall_only);
let cost = cost_usd_micros.map(|usd_micros| Cost {
usd_micros,
source: CostSource::Catalog,
});
Run {
id: run_id,
parent_id: None,
@ -1238,10 +1257,11 @@ mod runs {
completed_at: Some(created_at),
},
timing,
billing: total_usd_micros.map(|total_usd_micros| RunBillingSummary {
total_usd_micros: Some(total_usd_micros),
}),
size: RunSize::from_total_usd_micros(total_usd_micros),
usage: Usage {
tokens: TokenCounts::default(),
cost,
},
size: RunSize::from_cost(cost),
ask_fabro: Default::default(),
diff: None,
pull_request: None,
@ -1463,7 +1483,6 @@ mod runs {
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind,
FailoverContinuation, InputSource, McpToolSummary, SkillActivationSource, SkillSummary,
TokenUsage,
};
let run_id = demo_run_id(1);
@ -1508,13 +1527,11 @@ mod runs {
|model: &str, text: &str, input: u64, output: u64| CodingEvent::AssistantMessage {
text: text.into(),
model: model.into(),
usage: TokenUsage {
usage: Usage::from(TokenCounts {
input,
output,
..TokenUsage::default()
},
cost_usd_micros: None,
cost_source: None,
..TokenCounts::default()
}),
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -1650,19 +1667,18 @@ mod runs {
turns_used: 2,
}),
agent(CodingEvent::RouteFailover {
from: "anthropic/claude-opus-4.6".into(),
to: "openai/gpt-5.4".into(),
attempt: 1,
error: ErrorData::new(ErrorKind::Llm, "rate limited: retry after 30s"),
usage: TokenUsage {
from: "anthropic/claude-opus-4.6".into(),
to: "openai/gpt-5.4".into(),
attempt: 1,
error: ErrorData::new(ErrorKind::Llm, "rate limited: retry after 30s"),
usage: Usage::from(TokenCounts {
input: 3_600,
output: 540,
..TokenUsage::default()
},
cost_usd_micros: None,
inference_ms: 4_200,
tool_ms: 900,
continuation: FailoverContinuation::ContinueTurn,
..TokenCounts::default()
}),
inference_ms: 4_200,
tool_ms: 900,
continuation: FailoverContinuation::ContinueTurn,
}),
agent(CodingEvent::CompactionCompleted {
original_turn_count: 20,
@ -1670,12 +1686,11 @@ mod runs {
summary_token_estimate: 500,
tracked_file_count: 2,
reason: CompactionReason::Threshold,
usage: TokenUsage {
usage: Usage::from(TokenCounts {
input: 2_000,
output: 500,
..TokenUsage::default()
},
cost_usd_micros: None,
..TokenCounts::default()
}),
}),
call_started(
"write_file",
@ -1757,153 +1772,91 @@ mod runs {
projection
}
pub(super) fn billing() -> RunBilling {
RunBilling {
pub(super) fn usage() -> RunUsage {
RunUsage {
stages: vec![
RunBillingStage {
stage: BillingStageRef {
RunUsageStage {
stage: UsageStageRef {
id: "detect-drift".into(),
name: "Detect Drift".into(),
},
model: Some(billing_model(
model: Some(usage_model(
lithos_llm::catalog::builtin::anthropic(),
"claude-opus-4-6",
)),
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 12480,
output_tokens: 3210,
reasoning_tokens: 0,
total_tokens: 15690,
total_usd_micros: Some(480_000),
},
usage: priced(12480, 3210, 480_000),
timing: fabro_types::StageTiming::wall_only(72_000),
started_at: None,
state: Some(StageState::Succeeded),
},
RunBillingStage {
stage: BillingStageRef {
RunUsageStage {
stage: UsageStageRef {
id: "propose-changes".into(),
name: "Propose Changes".into(),
},
model: Some(billing_model(
model: Some(usage_model(
lithos_llm::catalog::builtin::gemini(),
"gemini-3.1-pro-preview",
)),
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 28640,
output_tokens: 8750,
reasoning_tokens: 0,
total_tokens: 37390,
total_usd_micros: Some(720_000),
},
usage: priced(28640, 8750, 720_000),
timing: fabro_types::StageTiming::wall_only(154_000),
started_at: None,
state: Some(StageState::Succeeded),
},
RunBillingStage {
stage: BillingStageRef {
RunUsageStage {
stage: UsageStageRef {
id: "review-changes".into(),
name: "Review Changes".into(),
},
model: Some(billing_model(
model: Some(usage_model(
lithos_llm::catalog::builtin::openai(),
"gpt-5.3-codex",
)),
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 9120,
output_tokens: 2640,
reasoning_tokens: 0,
total_tokens: 11760,
total_usd_micros: Some(190_000),
},
usage: priced(9120, 2640, 190_000),
timing: fabro_types::StageTiming::wall_only(45_000),
started_at: None,
state: Some(StageState::Succeeded),
},
RunBillingStage {
stage: BillingStageRef {
RunUsageStage {
stage: UsageStageRef {
id: "apply-changes".into(),
name: "Apply Changes".into(),
},
model: Some(billing_model(
model: Some(usage_model(
lithos_llm::catalog::builtin::anthropic(),
"claude-opus-4-6",
)),
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 21300,
output_tokens: 6480,
reasoning_tokens: 0,
total_tokens: 27780,
total_usd_micros: Some(870_000),
},
usage: priced(21300, 6480, 870_000),
timing: fabro_types::StageTiming::wall_only(118_000),
started_at: None,
state: Some(StageState::Running),
},
],
totals: RunBillingTotals {
cache_read_tokens: 0,
cache_write_tokens: 0,
timing: fabro_types::RunTiming::wall_only(389_000),
input_tokens: 71540,
output_tokens: 21080,
reasoning_tokens: 0,
total_tokens: 92620,
total_usd_micros: Some(2_260_000),
totals: RunUsageTotals {
timing: fabro_types::RunTiming::wall_only(389_000),
usage: priced(71540, 21080, 2_260_000),
},
by_model: vec![
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 33780,
output_tokens: 9690,
reasoning_tokens: 0,
total_tokens: 43470,
total_usd_micros: Some(1_350_000),
},
model: billing_model(
UsageByModel {
usage: priced(33780, 9690, 1_350_000),
model: usage_model(
lithos_llm::catalog::builtin::anthropic(),
"claude-opus-4-6",
),
stages: 2,
stages: 2,
},
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 28640,
output_tokens: 8750,
reasoning_tokens: 0,
total_tokens: 37390,
total_usd_micros: Some(720_000),
},
model: billing_model(
UsageByModel {
usage: priced(28640, 8750, 720_000),
model: usage_model(
lithos_llm::catalog::builtin::gemini(),
"gemini-3.1-pro-preview",
),
stages: 1,
stages: 1,
},
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 9120,
output_tokens: 2640,
reasoning_tokens: 0,
total_tokens: 11760,
total_usd_micros: Some(190_000),
},
model: billing_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"),
stages: 1,
UsageByModel {
usage: priced(9120, 2640, 190_000),
model: usage_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"),
stages: 1,
},
],
}
@ -2243,76 +2196,62 @@ mod workflows {
}
}
mod billing {
mod usage {
use fabro_api::types::*;
use lithos_llm::catalog::ProviderId;
use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage};
fn billing_model(provider: ProviderId, model_id: &str) -> BillingModelRef {
BillingModelRef {
fn usage_model(provider: ProviderId, model_id: &str) -> UsageModelRef {
UsageModelRef {
provider,
model_id: model_id.into(),
speed: None,
}
}
pub(super) fn aggregate() -> AggregateBilling {
AggregateBilling {
totals: AggregateBillingTotals {
cache_read_tokens: 0,
cache_write_tokens: 0,
runs: 9,
input_tokens: 643_860,
output_tokens: 189_720,
reasoning_tokens: 0,
timing: fabro_types::RunTiming::wall_only(3_501_000),
total_tokens: 833_580,
total_usd_micros: Some(20_340_000),
/// Demo usage priced from the catalog.
fn priced(input: u64, output: u64, usd_micros: u64) -> Usage {
Usage {
tokens: TokenCounts {
input,
output,
..TokenCounts::default()
},
cost: Some(Cost {
usd_micros,
source: CostSource::Catalog,
}),
}
}
pub(super) fn aggregate() -> AggregateUsage {
AggregateUsage {
totals: AggregateUsageTotals {
runs: 9,
timing: fabro_types::RunTiming::wall_only(3_501_000),
usage: priced(643_860, 189_720, 20_340_000),
},
by_model: vec![
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 304_020,
output_tokens: 87_210,
reasoning_tokens: 0,
total_tokens: 391_230,
total_usd_micros: Some(12_150_000),
},
model: billing_model(
UsageByModel {
usage: priced(304_020, 87_210, 12_150_000),
model: usage_model(
lithos_llm::catalog::builtin::anthropic(),
"claude-opus-4-6",
),
stages: 18,
stages: 18,
},
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 257_760,
output_tokens: 78_750,
reasoning_tokens: 0,
total_tokens: 336_510,
total_usd_micros: Some(6_480_000),
},
model: billing_model(
UsageByModel {
usage: priced(257_760, 78_750, 6_480_000),
model: usage_model(
lithos_llm::catalog::builtin::gemini(),
"gemini-3.1-pro-preview",
),
stages: 9,
stages: 9,
},
BillingByModel {
billing: BilledTokenCounts {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 82_080,
output_tokens: 23_760,
reasoning_tokens: 0,
total_tokens: 105_840,
total_usd_micros: Some(1_710_000),
},
model: billing_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"),
stages: 9,
UsageByModel {
usage: priced(82_080, 23_760, 1_710_000),
model: usage_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"),
stages: 9,
},
],
}

View file

@ -2308,7 +2308,7 @@ index 1111111..2222222 160000
failure: None,
final_git_commit_sha: None,
stages: Vec::new(),
billing: None,
usage: None,
total_retries: 0,
diff: fabro_types::RunDiff {
patch: Some(patch.to_string()),

View file

@ -12,6 +12,7 @@ use fabro_types::{
};
use httpmock::Method::{GET, POST};
use httpmock::{HttpMockRequest, HttpMockResponse, MockServer};
use lithos_llm::types::Usage;
use serde_json::json;
use tokio::fs;
#[expect(
@ -216,7 +217,7 @@ fn run_with_status(
completed_at: None,
},
timing: None,
billing: None,
usage: Usage::default(),
size: fabro_types::RunSize::default(),
ask_fabro: fabro_types::AskFabro::default(),
diff: None,

View file

@ -23,30 +23,30 @@ use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
use bytes::Bytes;
use chrono::{DateTime, Utc};
pub use fabro_api::types::{
AggregateBilling, AggregateBillingTotals, ApiQuestion, AppendEventResponse, ArtifactEntry,
AggregateUsage, AggregateUsageTotals, ApiQuestion, AppendEventResponse, ArtifactEntry,
ArtifactListResponse, BatchDeleteRunsRequest, BatchDeleteRunsResponse, BatchDeleteRunsResult,
BatchDeleteRunsResultOutcome, BatchDeleteRunsSummary, BatchRunLifecycleRequest,
BatchRunLifecycleResponse, BatchRunLifecycleResult, BatchRunLifecycleResultOutcome,
BatchRunLifecycleSummary, BillingByModel, BillingStageRef, CloseRunPullRequestResponse,
CompletionResponse, CompletionUsage, CreateCompletionRequest, CreateRunPullRequestRequest,
CreateSecretRequest, CreateVariableRequest, DeleteRunResponse, DeleteRunSandbox,
DeleteSecretRequest, DenyRunRequest, DiskUsageResponse, DiskUsageRunRow, DiskUsageSummaryRow,
ErrorResponseEntry, ForkRequest, ForkResponse, IntegrationConnectionKind,
IntegrationConnectionState, IntegrationConnectionStatus, IntegrationProvider,
IntegrationStatus, LinkRunPullRequestRequest, MergeRunPullRequestRequest,
MergeRunPullRequestResponse, ModelReference, PaginatedEventList, PaginatedRunList,
PaginationMeta, PreflightResponse, PreviewUrlRequest, PreviewUrlResponse, Provider,
ProviderCredentialTestRequest, ProviderCredentialTestResponse, ProviderList, PruneRunEntry,
PruneRunsRequest, PruneRunsResponse, RenderWorkflowGraphDirection, RenderWorkflowGraphRequest,
RewindRequest, RewindResponse, Run, RunArtifactEntry, RunArtifactListResponse, RunBilling,
RunBillingStage, RunBillingTotals, RunError, RunManifest, RunStage, SandboxDetails,
SandboxFileEntry, SandboxFileListResponse, SandboxService, SandboxServiceListResponse,
SshAccessRequest, SshAccessResponse, StageHandler, StageState, StartRunRequest,
SubmitAnswerRequest, SystemCpuResourceScope, SystemCpuResources, SystemDiskResourceScope,
SystemDiskResources, SystemInfoResponse, SystemIntegrationStatus, SystemIntegrationsResponse,
SystemMemoryResourceScope, SystemMemoryResources, SystemRepairRunIssue,
SystemRepairRunsResponse, SystemResourcesResponse, SystemRunCounts, TimelineEntryResponse,
UpdateVariableRequest, VariableListResponse, VncPreviewResponse, WriteBlobResponse,
BatchRunLifecycleSummary, CloseRunPullRequestResponse, CompletionResponse,
CreateCompletionRequest, CreateRunPullRequestRequest, CreateSecretRequest,
CreateVariableRequest, DeleteRunResponse, DeleteRunSandbox, DeleteSecretRequest,
DenyRunRequest, DiskUsageResponse, DiskUsageRunRow, DiskUsageSummaryRow, ErrorResponseEntry,
ForkRequest, ForkResponse, IntegrationConnectionKind, IntegrationConnectionState,
IntegrationConnectionStatus, IntegrationProvider, IntegrationStatus, LinkRunPullRequestRequest,
MergeRunPullRequestRequest, MergeRunPullRequestResponse, ModelReference, PaginatedEventList,
PaginatedRunList, PaginationMeta, PreflightResponse, PreviewUrlRequest, PreviewUrlResponse,
Provider, ProviderCredentialTestRequest, ProviderCredentialTestResponse, ProviderList,
PruneRunEntry, PruneRunsRequest, PruneRunsResponse, RenderWorkflowGraphDirection,
RenderWorkflowGraphRequest, RewindRequest, RewindResponse, Run, RunArtifactEntry,
RunArtifactListResponse, RunError, RunManifest, RunStage, RunUsage, RunUsageStage,
RunUsageTotals, SandboxDetails, SandboxFileEntry, SandboxFileListResponse, SandboxService,
SandboxServiceListResponse, SshAccessRequest, SshAccessResponse, StageHandler, StageState,
StartRunRequest, SubmitAnswerRequest, SystemCpuResourceScope, SystemCpuResources,
SystemDiskResourceScope, SystemDiskResources, SystemInfoResponse, SystemIntegrationStatus,
SystemIntegrationsResponse, SystemMemoryResourceScope, SystemMemoryResources,
SystemRepairRunIssue, SystemRepairRunsResponse, SystemResourcesResponse, SystemRunCounts,
TimelineEntryResponse, UpdateVariableRequest, UsageByModel, UsageStageRef,
VariableListResponse, VncPreviewResponse, WriteBlobResponse,
};
use fabro_auth::SqlVaultCredentialSource;
use fabro_automation::{self, AutomationStore};
@ -88,7 +88,7 @@ use fabro_types::settings::server::{
GithubIntegrationSettings, GithubIntegrationStrategy, LogDestination,
};
use fabro_types::{
AgentBackend, AskFabro, AskFabroUnavailableReason, BilledTokenCounts, BlobHash, EventBody,
AgentBackend, AskFabro, AskFabroUnavailableReason, BlobHash, EventBody,
InterviewQuestionRecord, ModelRef, ModelTestMode, PairId, PairMessageId, PairTarget,
PendingReason, Principal, PullRequestLink, QuestionType, RunControlAction, RunEvent, RunId,
RunRunnableSource, RunStatusKind, SandboxProviderKind, ServerSettings, SessionCapability,
@ -113,6 +113,7 @@ use fabro_workflow::run_status::{FailureReason, RunStatus, SuccessReason};
use fabro_workflow::{Error as WorkflowError, operations, pull_request};
use futures_util::future::join_all;
use lithos_llm::catalog::ProviderId;
use lithos_llm::types::Usage;
use sha2::{Digest, Sha256};
use tempfile::NamedTempFile;
use tokio::fs;
@ -317,19 +318,19 @@ enum ExecutionResult {
const WORKER_CANCEL_GRACE: Duration = Duration::from_secs(5);
const TERMINAL_DELETE_WORKER_GRACE: Duration = Duration::from_millis(50);
const WORKER_CONTROL_ENQUEUE_TIMEOUT: Duration = Duration::from_secs(1);
/// Per-model billing totals.
/// Per-model usage totals.
#[derive(Default)]
struct ModelBillingTotals {
stages: i64,
billing: BilledTokenCounts,
pub(crate) struct ModelUsageTotals {
pub(crate) stages: i64,
pub(crate) usage: Usage,
}
/// In-memory aggregate billing counters, reset on server restart.
/// In-memory aggregate usage counters, reset on server restart.
#[derive(Default)]
struct BillingAccumulator {
total_runs: i64,
total_timing: fabro_types::RunTiming,
by_model: HashMap<ModelRef, ModelBillingTotals>,
pub(crate) struct UsageAccumulator {
pub(crate) total_runs: i64,
pub(crate) total_timing: fabro_types::RunTiming,
pub(crate) by_model: HashMap<ModelRef, ModelUsageTotals>,
}
pub(crate) type RegistryFactoryOverride =
@ -1098,7 +1099,7 @@ fn resolve_slack_lifecycle_route_channel(
/// Shared application state for the server.
pub struct AppState {
runs: Mutex<HashMap<RunId, ManagedRun>>,
aggregate_billing: Mutex<BillingAccumulator>,
aggregate_usage: Mutex<UsageAccumulator>,
pub(crate) stores: AppStores,
session_runtimes: SessionRuntimeManager,
artifact_store: ArtifactStore,
@ -1296,16 +1297,16 @@ pub(crate) struct ResolvedAppStateSettings {
pub(crate) llm_overlay: LlmLayer,
}
fn accumulate_billing_rollup(
accumulator: &mut BillingAccumulator,
rollup: &fabro_workflow::ProjectionBillingRollup,
fn accumulate_usage_rollup(
accumulator: &mut UsageAccumulator,
rollup: &fabro_workflow::ProjectionUsageRollup,
) {
accumulator.total_runs += 1;
accumulator.total_timing = accumulator.total_timing.saturating_add(&rollup.timing);
for model in &rollup.by_model {
let entry = accumulator.by_model.entry(model.model.clone()).or_default();
entry.stages += model.stages;
entry.billing.add_counts(&model.billing);
entry.usage = entry.usage.saturating_add(model.usage);
}
}
@ -2558,7 +2559,7 @@ pub(crate) fn build_app_state(config: AppStateConfig) -> anyhow::Result<Arc<AppS
};
Ok(Arc::new(AppState {
runs: Mutex::new(HashMap::new()),
aggregate_billing: Mutex::new(BillingAccumulator::default()),
aggregate_usage: Mutex::new(UsageAccumulator::default()),
stores: AppStores {
runs: store,
run_summaries,
@ -4228,12 +4229,12 @@ async fn execute_run_in_process(state: Arc<AppState>, run_id: RunId) {
if let Some(ref projection) = final_projection {
if projection.current_checkpoint().is_some() {
let mut agg = state
.aggregate_billing
.aggregate_usage
.lock()
.expect("aggregate_billing lock poisoned");
accumulate_billing_rollup(
.expect("aggregate_usage lock poisoned");
accumulate_usage_rollup(
&mut agg,
&fabro_workflow::billing_rollup_from_projection(projection),
&fabro_workflow::usage_rollup_from_projection(projection),
);
}
}
@ -4473,12 +4474,12 @@ async fn execute_run_subprocess(state: Arc<AppState>, run_id: RunId) {
if final_state.current_checkpoint().is_some() {
let mut agg = state
.aggregate_billing
.aggregate_usage
.lock()
.expect("aggregate_billing lock poisoned");
accumulate_billing_rollup(
.expect("aggregate_usage lock poisoned");
accumulate_usage_rollup(
&mut agg,
&fabro_workflow::billing_rollup_from_projection(&final_state),
&fabro_workflow::usage_rollup_from_projection(&final_state),
);
}

View file

@ -9,7 +9,6 @@ use super::{ApiError, AppState, IntoResponse, Json, Response, StatusCode, demo};
mod artifacts;
pub(in crate::server) mod automations;
mod billing;
mod completions;
mod environments;
pub(in crate::server) mod events;
@ -27,6 +26,7 @@ mod secrets;
mod sessions;
mod steer;
pub(in crate::server) mod system;
mod usage;
mod variables;
mod worker_control;
mod workflow_versions;
@ -135,7 +135,7 @@ pub(super) fn demo_routes() -> Router<Arc<AppState>> {
"/runs/{id}/stages/{stageId}/artifacts/download",
get(not_implemented),
)
.route("/runs/{id}/billing", get(demo::get_run_billing))
.route("/runs/{id}/usage", get(demo::get_run_usage))
.route("/runs/{id}/settings", get(demo::get_run_settings))
.route("/runs/{id}/preview", post(demo::generate_preview_url_stub))
.route("/runs/{id}/ssh", post(demo::create_ssh_access_stub))
@ -178,7 +178,7 @@ pub(super) fn demo_routes() -> Router<Arc<AppState>> {
.route("/system/df", get(demo::get_system_disk_usage))
.route("/system/repair/runs", get(demo::get_system_repair_runs))
.route("/system/prune/runs", post(demo::prune_runs))
.route("/billing", get(demo::get_aggregate_billing))
.route("/usage", get(demo::get_aggregate_usage))
.route("/workflows", get(demo::list_workflows))
.route("/workflows/{name}", get(demo::get_workflow))
.route("/workflows/{name}/runs", get(demo::list_workflow_runs))
@ -208,7 +208,7 @@ pub(super) fn real_routes() -> Router<Arc<AppState>> {
.route("/insights/history", get(not_implemented))
.merge(runs::routes())
.merge(events::routes())
.merge(billing::routes())
.merge(usage::routes())
.merge(pull_requests::routes())
.merge(artifacts::routes())
.merge(automations::routes())

View file

@ -873,7 +873,7 @@ mod tests {
fixtures, test_support,
};
use fabro_workflow::event as workflow_event;
use pebble_coding_agent::events::{CodingAgentEvent, TokenUsage};
use pebble_coding_agent::events::{CodingAgentEvent, Usage};
use tower::ServiceExt;
use super::*;
@ -908,9 +908,7 @@ mod tests {
CodingEvent::AssistantMessage {
text: "I found the issue.".to_string(),
model: "gpt-5.4".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -945,9 +943,7 @@ mod tests {
CodingEvent::AssistantMessage {
text: "wrong stage".to_string(),
model: "gpt-5.4".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 0,
context_window: None,
reasoning: None,

View file

@ -1973,6 +1973,7 @@ mod resume_tests {
use fabro_static::EnvVars;
use fabro_test::{TwinScenario, TwinScenarios, twin_openai};
use fabro_types::{RunId, SessionId};
use pebble_coding_agent::state::SESSION_RECORD_FORMAT_VERSION;
use tower::ServiceExt;
use crate::server::{AppState, spawn_scheduler};
@ -2122,6 +2123,115 @@ mod resume_tests {
events
}
/// A record written by an older build is refused, not read: pebble checks
/// the format version before it reads the route, and fabro turns that
/// refusal into a turn failure naming both versions. Old runs get no
/// migration, so this is what a session persisted before the record
/// format moved sees on its next turn.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn a_stored_record_in_an_older_format_fails_the_turn_naming_both_versions() {
let twin = twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
TwinScenarios::new(namespace.clone())
.scenario(
TwinScenario::responses(MODEL)
.input_contains("First question")
.text("First answer"),
)
.load(twin)
.await;
let state = twin_backed_state(twin.base_url.clone(), &namespace);
spawn_scheduler(Arc::clone(&state));
let app = build_test_router(Arc::clone(&state));
let workspace = tempfile::tempdir().unwrap();
let run_id = completed_run(&app, workspace.path()).await;
let created = json_response(
&app,
post_json(
&format!("/runs/{run_id}/sessions"),
&serde_json::json!({ "title": "Ask Fabro", "model": MODEL }),
),
StatusCode::CREATED,
)
.await;
let session_id: SessionId = created["id"].as_str().unwrap().parse().unwrap();
turn(&app, session_id, "First question").await;
let stored = state
.stores
.session_records
.get(session_id)
.await
.unwrap()
.expect("the first turn persists the record");
// The record as an older build wrote it: the previous format version.
// The store parses it, and the version check refuses it on resume.
let previous = SESSION_RECORD_FORMAT_VERSION - 1;
let mut older = stored.record.clone();
older.format_version = previous;
state
.stores
.session_records
.put(session_id, run_id, &older, chrono::Utc::now())
.await
.unwrap();
state
.session_runtimes()
.load_or_create_runtime(session_id)
.clear_agent()
.await;
let response = app
.clone()
.oneshot(post_json(
&format!("/sessions/{session_id}/turns"),
&serde_json::json!({ "input": "Second question" }),
))
.await
.unwrap();
assert_eq!(response.status(), StatusCode::OK);
let bytes = to_bytes(response.into_body(), usize::MAX).await.unwrap();
let body = String::from_utf8(bytes.to_vec()).unwrap();
let events: Vec<serde_json::Value> = body
.lines()
.filter_map(|line| line.strip_prefix("data: "))
.map(|data| serde_json::from_str(data).unwrap())
.collect();
let failed = events
.iter()
.find(|event| event["event"] == "run.session.turn.failed")
.unwrap_or_else(|| panic!("the resumed turn fails: {events:#?}"));
assert_eq!(failed["properties"]["code"], "agent_error");
assert_eq!(failed["properties"]["retryable"], false);
let error = failed["properties"]["error"].as_str().unwrap();
let expected = format!(
"session record format version {previous} is not supported (this build requires {SESSION_RECORD_FORMAT_VERSION})"
);
assert!(
error.contains(&expected),
"the failure names both versions: {error}"
);
assert!(
!events
.iter()
.any(|event| event["event"] == "run.session.turn.succeeded"),
"a refused record runs no turn: {events:#?}"
);
// The stored record is untouched, so a build that reads its format
// can still resume it.
let after = state
.stores
.session_records
.get(session_id)
.await
.unwrap()
.expect("the refused record stays stored");
assert_eq!(after.record.format_version, previous);
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn a_resumed_session_continues_its_conversation_past_the_event_log() {
let twin = twin_openai().await;

View file

@ -10,18 +10,19 @@ use fabro_slack::config::{
};
use fabro_static::EnvVars;
use fabro_types::settings::server::GithubIntegrationSettings;
use fabro_types::sum_usage;
use fabro_vault::Vault;
use tokio::time::timeout;
use super::super::{
AggregateBilling, AggregateBillingTotals, ApiError, AppState, BilledTokenCounts,
BillingByModel, DfParams, FABRO_VERSION, GithubIntegrationStrategy, IntegrationConnectionState,
IntegrationProvider, IntegrationStatus, IntoResponse, Json, Path, PruneRunsRequest,
PruneRunsResponse, Query, RequiredUser, Response, Router, RunStatus, State, StatusCode,
SystemInfoResponse, SystemIntegrationStatus, SystemIntegrationsResponse, SystemRepairRunIssue,
SystemRepairRunsResponse, SystemRunCounts, build_disk_usage_response, build_prune_plan,
counts_toward_scheduler_capacity, delete_run_internal, diagnostics, get, post,
resource_sampler, spawn_blocking, system_sandbox_provider, to_i64,
AggregateUsage, AggregateUsageTotals, ApiError, AppState, DfParams, FABRO_VERSION,
GithubIntegrationStrategy, IntegrationConnectionState, IntegrationProvider, IntegrationStatus,
IntoResponse, Json, Path, PruneRunsRequest, PruneRunsResponse, Query, RequiredUser, Response,
Router, RunStatus, State, StatusCode, SystemInfoResponse, SystemIntegrationStatus,
SystemIntegrationsResponse, SystemRepairRunIssue, SystemRepairRunsResponse, SystemRunCounts,
UsageByModel, build_disk_usage_response, build_prune_plan, counts_toward_scheduler_capacity,
delete_run_internal, diagnostics, get, post, resource_sampler, spawn_blocking,
system_sandbox_provider, to_i64,
};
const SERVER_DIAGNOSTICS_TIMEOUT: Duration = Duration::from_secs(25);
@ -38,7 +39,7 @@ pub(super) fn routes() -> Router<Arc<AppState>> {
.route("/system/df", get(get_system_df))
.route("/system/repair/runs", get(get_system_repair_runs))
.route("/system/prune/runs", post(prune_runs))
.route("/billing", get(get_aggregate_billing))
.route("/usage", get(get_aggregate_usage))
}
pub(in crate::server) async fn health() -> Response {
@ -715,41 +716,26 @@ pub(in crate::server) async fn openapi_spec() -> Response {
Json(value).into_response()
}
async fn get_aggregate_billing(
_auth: RequiredUser,
State(state): State<Arc<AppState>>,
) -> Response {
async fn get_aggregate_usage(_auth: RequiredUser, State(state): State<Arc<AppState>>) -> Response {
let agg = state
.aggregate_billing
.aggregate_usage
.lock()
.expect("aggregate_billing lock poisoned");
let by_model: Vec<BillingByModel> = agg
.expect("aggregate_usage lock poisoned");
let by_model: Vec<UsageByModel> = agg
.by_model
.iter()
.map(|(model, totals)| BillingByModel {
billing: totals.billing.clone(),
model: model.clone(),
stages: totals.stages,
.map(|(model, totals)| UsageByModel {
model: model.clone(),
stages: totals.stages,
usage: totals.usage,
})
.collect();
let total_billing =
agg.by_model
.values()
.fold(BilledTokenCounts::default(), |mut acc, totals| {
acc.add_counts(&totals.billing);
acc
});
let response = AggregateBilling {
totals: AggregateBillingTotals {
cache_read_tokens: total_billing.cache_read_tokens,
cache_write_tokens: total_billing.cache_write_tokens,
input_tokens: total_billing.input_tokens,
output_tokens: total_billing.output_tokens,
reasoning_tokens: total_billing.reasoning_tokens,
runs: agg.total_runs,
timing: agg.total_timing,
total_tokens: total_billing.total_tokens,
total_usd_micros: total_billing.total_usd_micros,
let usage = sum_usage(agg.by_model.values().map(|totals| totals.usage));
let response = AggregateUsage {
totals: AggregateUsageTotals {
runs: agg.total_runs,
timing: agg.total_timing,
usage,
},
by_model,
};

View file

@ -4,18 +4,19 @@ use std::sync::Arc;
use chrono::{DateTime, Utc};
use fabro_types::{
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
usage_is_empty,
};
use super::super::{
AppState, BillingByModel, BillingStageRef, IntoResponse, Json, ListResponse, PaginationParams,
Path, Query, RequiredUser, Response, Router, RunBilling, RunBillingStage, RunBillingTotals,
RunId, RunStage, State, StatusCode, get, parse_run_id_path,
AppState, IntoResponse, Json, ListResponse, PaginationParams, Path, Query, RequiredUser,
Response, Router, RunId, RunStage, RunUsage, RunUsageStage, RunUsageTotals, State, StatusCode,
UsageByModel, UsageStageRef, get, parse_run_id_path,
};
pub(super) fn routes() -> Router<Arc<AppState>> {
Router::new()
.route("/runs/{id}/stages", get(list_run_stages))
.route("/runs/{id}/billing", get(get_run_billing))
.route("/runs/{id}/usage", get(get_run_usage))
}
fn run_stage_from_projection(
@ -41,7 +42,7 @@ fn run_stage_from_projection(
id: stage_id.clone(),
name: stage_id.node_id().to_owned(),
handler,
billing: stage.usage.clone(),
usage: stage.usage,
status: stage.effective_state(),
wall_time_ms: stage.live_wall_time_ms(now),
node_id: stage_id.node_id().to_owned(),
@ -82,7 +83,7 @@ async fn list_run_stages(
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()
}
async fn get_run_billing(
async fn get_run_usage(
_auth: RequiredUser,
State(state): State<Arc<AppState>>,
Path(id): Path<RunId>,
@ -92,14 +93,14 @@ async fn get_run_billing(
Err(err) => return err.into_response(),
};
let rollup = fabro_workflow::billing_rollup_from_projection(&projection);
let rollup = fabro_workflow::usage_rollup_from_projection(&projection);
let by_model = rollup
.by_model
.iter()
.map(|model| BillingByModel {
billing: model.billing.clone(),
model: model.model.clone(),
stages: model.stages,
.map(|model| UsageByModel {
model: model.model.clone(),
stages: model.stages,
usage: model.usage,
})
.collect::<Vec<_>>();
@ -108,7 +109,7 @@ async fn get_run_billing(
.iter()
.map(|stage| (stage.node_id.as_str(), stage))
.collect::<HashMap<_, _>>();
let live_rows = live_billing_rows(&projection, Utc::now());
let live_rows = live_usage_rows(&projection, Utc::now());
let totals_timing = live_rows.iter().fold(StageTiming::default(), |acc, row| {
acc.saturating_add(&row.timing)
});
@ -116,13 +117,11 @@ async fn get_run_billing(
.into_iter()
.map(|row| {
let rollup_stage = rollup_by_node.get(row.node_id.as_str());
RunBillingStage {
billing: rollup_stage
.map(|stage| stage.billing.clone())
.unwrap_or_default(),
RunUsageStage {
usage: rollup_stage.map(|stage| stage.usage).unwrap_or_default(),
model: rollup_stage.and_then(|stage| stage.model.as_ref()).cloned(),
timing: row.timing,
stage: BillingStageRef {
stage: UsageStageRef {
id: row.node_id.clone(),
name: row.node_id,
},
@ -132,25 +131,19 @@ async fn get_run_billing(
})
.collect::<Vec<_>>();
let response = RunBilling {
let response = RunUsage {
by_model,
stages,
totals: RunBillingTotals {
cache_read_tokens: rollup.totals.cache_read_tokens,
cache_write_tokens: rollup.totals.cache_write_tokens,
input_tokens: rollup.totals.input_tokens,
output_tokens: rollup.totals.output_tokens,
reasoning_tokens: rollup.totals.reasoning_tokens,
timing: totals_timing.into(),
total_tokens: rollup.totals.total_tokens,
total_usd_micros: rollup.totals.total_usd_micros,
totals: RunUsageTotals {
timing: totals_timing.into(),
usage: rollup.totals,
},
};
(StatusCode::OK, Json(response)).into_response()
}
struct LiveBillingRow {
struct LiveUsageRow {
node_id: String,
timing: StageTiming,
started_at: Option<DateTime<Utc>>,
@ -158,19 +151,19 @@ struct LiveBillingRow {
latest_visit: u32,
}
fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<LiveBillingRow> {
fn live_usage_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<LiveUsageRow> {
let mut row_indices = HashMap::<String, usize>::new();
let mut rows = Vec::<LiveBillingRow>::new();
let mut rows = Vec::<LiveUsageRow>::new();
for (stage_id, stage) in projection.iter_stages() {
let node_id = stage_id.node_id();
if projection.is_boundary_stage(node_id) || !stage_has_billing_row(stage) {
if projection.is_boundary_stage(node_id) || !stage_has_usage_row(stage) {
continue;
}
let index = *row_indices.entry(node_id.to_string()).or_insert_with(|| {
let index = rows.len();
rows.push(LiveBillingRow {
rows.push(LiveUsageRow {
node_id: node_id.to_string(),
timing: StageTiming::default(),
started_at: None,
@ -193,9 +186,9 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
rows
}
fn stage_has_billing_row(stage: &StageProjection) -> bool {
fn stage_has_usage_row(stage: &StageProjection) -> bool {
stage.completion.is_some()
|| stage.timing.is_some()
|| !stage.usage.is_zero()
|| !usage_is_empty(&stage.usage)
|| stage.started_at.is_some()
}

File diff suppressed because it is too large Load diff

View file

@ -114,11 +114,10 @@ async fn append_completed_run_with_final_patch(
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: Some("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb".to_string()),
final_patch: Some(final_patch.to_string()),
diff_summary: None,
billing: None,
usage: None,
},
)
.await

View file

@ -26,7 +26,7 @@ const WAIT_DOT: &str = r#"digraph Test {
}"#;
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn aggregate_billing_increments_after_run_completes() {
async fn aggregate_usage_increments_after_run_completes() {
let workspace = tempfile::tempdir().unwrap();
let state = test_app_state_with_options(test_settings(), 5);
let app = test_app_with_scheduler(state);
@ -45,7 +45,7 @@ async fn aggregate_billing_increments_after_run_completes() {
for _ in 0..POLL_ATTEMPTS {
let req = Request::builder()
.method("GET")
.uri(api("/billing"))
.uri(api("/usage"))
.body(Body::empty())
.unwrap();
@ -66,7 +66,7 @@ async fn aggregate_billing_increments_after_run_completes() {
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn run_billing_includes_completed_non_llm_stages() {
async fn run_usage_includes_completed_non_llm_stages() {
let workspace = tempfile::tempdir().unwrap();
let state = test_app_state_with_options(test_settings(), 5);
let app = test_app_with_scheduler(state);
@ -80,12 +80,12 @@ async fn run_billing_includes_completed_non_llm_stages() {
let status = wait_for_run_status(&app, &run_id, &["succeeded", "failed"]).await;
assert_eq!(status, "succeeded");
let billing = run_billing(&app, &run_id).await;
assert_non_llm_billing(&billing, &["wait_task"]);
let usage = run_usage(&app, &run_id).await;
assert_non_llm_usage(&usage, &["wait_task"]);
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn run_billing_includes_completed_command_stages() {
async fn run_usage_includes_completed_command_stages() {
let workspace = tempfile::tempdir().unwrap();
let state = test_app_state_with_options(test_settings(), 5);
let app = test_app_with_scheduler(state);
@ -99,30 +99,30 @@ async fn run_billing_includes_completed_command_stages() {
let status = wait_for_run_status(&app, &run_id, &["succeeded", "failed"]).await;
assert_eq!(status, "succeeded");
let billing = run_billing(&app, &run_id).await;
assert_non_llm_billing(&billing, &["echo_task"]);
let usage = run_usage(&app, &run_id).await;
assert_non_llm_usage(&usage, &["echo_task"]);
}
async fn run_billing(app: &axum::Router, run_id: &str) -> serde_json::Value {
async fn run_usage(app: &axum::Router, run_id: &str) -> serde_json::Value {
let req = Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/billing")))
.uri(api(&format!("/runs/{run_id}/usage")))
.body(Body::empty())
.expect("run billing request should build");
.expect("run usage request should build");
let response = app.clone().oneshot(req).await.unwrap();
crate::helpers::response_json(
response,
StatusCode::OK,
format!("GET /api/v1/runs/{run_id}/billing"),
format!("GET /api/v1/runs/{run_id}/usage"),
)
.await
}
fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&str]) {
let stages = billing["stages"]
fn assert_non_llm_usage(usage: &serde_json::Value, expected_stage_ids: &[&str]) {
let stages = usage["stages"]
.as_array()
.expect("billing response should include stages");
.expect("usage response should include stages");
let mut stage_ids = stages
.iter()
.map(|stage| {
@ -138,10 +138,10 @@ fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&st
assert!(
stages.iter().all(|stage| {
stage["model"].is_null()
&& stage["billing"]["input_tokens"] == 0
&& stage["billing"]["output_tokens"] == 0
&& stage["billing"]["reasoning_tokens"] == 0
&& stage["billing"]["total_usd_micros"].is_null()
&& stage["usage"]["tokens"]["input"] == 0
&& stage["usage"]["tokens"]["output"] == 0
&& stage["usage"]["tokens"]["reasoning"] == 0
&& stage["usage"].get("cost").is_none()
}),
"every non-LLM stage should have null model and zero token counts: {stages:?}"
);
@ -156,17 +156,17 @@ fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&st
.sum();
assert_eq!(
billing["by_model"]
usage["by_model"]
.as_array()
.expect("billing response should include by_model")
.expect("usage response should include by_model")
.len(),
0
);
assert_eq!(billing["totals"]["input_tokens"], 0);
assert_eq!(billing["totals"]["output_tokens"], 0);
assert!(billing["totals"]["total_usd_micros"].is_null());
assert_eq!(usage["totals"]["usage"]["tokens"]["input"], 0);
assert_eq!(usage["totals"]["usage"]["tokens"]["output"], 0);
assert!(usage["totals"]["usage"].get("cost").is_none());
let total_wall_time_ms = billing["totals"]["timing"]["wall_time_ms"]
let total_wall_time_ms = usage["totals"]["timing"]["wall_time_ms"]
.as_u64()
.expect("totals should include timing.wall_time_ms");
assert_eq!(

View file

@ -558,7 +558,7 @@ mod tests {
failure: None,
final_git_commit_sha: Some("abc123".to_string()),
stages: Vec::new(),
billing: None,
usage: None,
total_retries: 0,
diff: RunDiff::default(),
});

View file

@ -0,0 +1,116 @@
{
"format_version": 4,
"scope": {
"session_id": "ses_root",
"root_session_id": "ses_root",
"parent_session_id": null,
"depth": 0
},
"provider": "anthropic",
"model": "claude-sonnet-5",
"created_at": "2026-01-01T00:00:00.500Z",
"updated_at": "2026-01-01T00:00:00.500Z",
"last_event_seq": 41,
"messages": [
{
"kind": "user",
"content": [
{
"type": "text",
"text": "read the crate root"
}
],
"timestamp": "2026-01-01T00:00:00.500Z"
},
{
"kind": "assistant",
"content": "Reading it now.",
"tool_calls": [
{
"id": "call_1",
"name": "read_file",
"input": {
"type": "function",
"value": "{\"path\":\"src/lib.rs\"}"
},
"provider_metadata": {
"openai": {
"item_id": "fc_1"
}
}
}
],
"provider_parts": [
{
"type": "reasoning",
"text": "the root is a facade",
"signature": "sig_1",
"signature_origin": "anthropic"
},
{
"type": "opaque",
"kind": "openai.reasoning",
"data": {
"id": "rs_1"
}
}
],
"usage": {
"input": 1200,
"output": 340,
"reasoning": 96,
"cache_read": 800,
"cache_write": 64
},
"response_id": "resp_1",
"timestamp": "2026-01-01T00:00:00.500Z"
},
{
"kind": "tool_results",
"results": [
{
"tool_call_id": "call_1",
"name": "read_file",
"content": [
{
"type": "text",
"text": "//! Pebble is a coding-agent loop library."
}
],
"is_error": false
}
],
"timestamp": "2026-01-01T00:00:00.500Z"
},
{
"kind": "compaction",
"summary": "[Context Summary]\nThe session read the crate root.",
"reason": "manual",
"original_turn_count": 8,
"preserved_turn_count": 2,
"estimated_tokens_before": 12000,
"summary_token_estimate": 24,
"tracked_file_count": 1,
"summary_truncated": false,
"usage": {
"input": 1200,
"output": 340,
"reasoning": 96,
"cache_read": 800,
"cache_write": 64
},
"cost_usd_micros": 1250,
"timestamp": "2026-01-01T00:00:00.500Z"
},
{
"kind": "steering",
"content": [
{
"type": "text",
"text": "also update the changelog"
}
],
"timestamp": "2026-01-01T00:00:00.500Z"
}
]
}

View file

@ -110,6 +110,7 @@ mod tests {
use chrono::TimeZone;
use fabro_types::fixtures;
use pebble_coding_agent::SessionScope;
use pebble_coding_agent::state::SESSION_RECORD_FORMAT_VERSION;
use super::*;
use crate::test_support;
@ -149,6 +150,44 @@ mod tests {
assert_eq!(stored.updated_at, updated_at);
}
/// A record an older build wrote is read back as it was stored: the
/// store parses the previous format, and pebble's version check, not a
/// parse error, is what refuses it on resume. Old runs get no migration.
#[tokio::test]
async fn get_reads_a_record_in_the_previous_format_for_pebble_to_refuse() {
const PREVIOUS_RECORD: &str = include_str!("fixtures/session_record_v4.json");
let pool =
test_support::in_memory_pool_with(&[fabro_db::RUN_SESSION_RECORDS_MIGRATION_SQL]);
let store = RunSessionRecordStore::new(pool.clone());
let session_id = SessionId::new();
sqlx::query(
"INSERT INTO run_session_records (session_id, run_id, record_json, updated_at_ms) \
VALUES (?, ?, ?, ?)",
)
.bind(session_id.to_string())
.bind(fixtures::RUN_1.to_string())
.bind(PREVIOUS_RECORD)
.bind(1_789_156_874_678_i64)
.execute(&pool)
.await
.unwrap();
let stored = store.get(session_id).await.unwrap().expect("stored record");
assert_eq!(stored.run_id, fixtures::RUN_1);
assert_eq!(
stored.record.format_version,
SESSION_RECORD_FORMAT_VERSION - 1,
"the fixture is the previous format"
);
assert!(
!stored.record.is_supported(),
"pebble refuses the previous format on resume rather than reading it"
);
assert_eq!(stored.record.provider.as_deref(), Some("anthropic"));
assert_eq!(stored.record.model.as_deref(), Some("claude-sonnet-5"));
}
#[tokio::test]
async fn put_replaces_an_earlier_record() {
let store = RunSessionRecordStore::new(test_support::in_memory_pool_with(&[

View file

@ -9,20 +9,19 @@ use fabro_types::run_event::{
};
use fabro_types::settings::run::RunEnvironmentSettings;
use fabro_types::{
AskFabro, BilledModelUsage, BilledTokenCounts, Checkpoint, CheckpointRecord,
CommandTermination, Conclusion, EventBody, FailureCategory, FailureSignature,
InterviewQuestionRecord, ModelRef, Outcome, PendingInterviewRecord, PendingReason,
PullRequestCreation, PullRequestCreationStatus, PullRequestLink, RepositoryRef, Run,
RunApproval, RunApprovalState, RunBillingSummary, RunControlAction, RunDiff, RunEvent, RunId,
RunLifecycle, RunLinks, RunModel, RunOrigin, RunProjection, RunSandbox, RunSandboxFailure,
RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, RunSize, RunSpec, RunStatus,
RunTimestamps, SandboxProviderKind, StageCompletion, StageHandler, StageId,
AskFabro, Checkpoint, CheckpointRecord, CommandTermination, Conclusion, EventBody,
FailureCategory, FailureSignature, InterviewQuestionRecord, ModelRef, ModelUsage, Outcome,
PendingInterviewRecord, PendingReason, PullRequestCreation, PullRequestCreationStatus,
PullRequestLink, RepositoryRef, Run, RunApproval, RunApprovalState, RunControlAction, RunDiff,
RunEvent, RunId, RunLifecycle, RunLinks, RunModel, RunOrigin, RunProjection, RunSandbox,
RunSandboxFailure, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, RunSize, RunSpec,
RunStatus, RunTimestamps, SandboxProviderKind, StageCompletion, StageHandler, StageId,
StageInferenceProjection, StageModelUsage, StageOutcome, StageProjection, StageState,
StartRecord, WorkflowRef, billing_rollup, first_event_seq, timing,
StartRecord, WorkflowRef, first_event_seq, sum_usage, timing, usage_rollup,
};
use fabro_util::error::render_compact_with_causes;
use lithos_llm::catalog::{ModelId, ProviderId};
use lithos_llm::types::TokenCounts;
use lithos_llm::types::Usage;
use pebble_coding_agent::events::CodingEvent;
use pebble_coding_agent::projection::SessionProjection;
@ -501,9 +500,9 @@ impl RunProjectionReducer for RunProjection {
return Ok(());
};
stage.response = Some(props.response.clone());
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
if let Some(usage) = &props.usage {
stage.usage = usage.usage;
stage.model = Some(usage.model().clone());
}
}
EventBody::StageCompleted(props) => {
@ -518,11 +517,11 @@ impl RunProjectionReducer for RunProjection {
stage.response = response;
stage.completion = Some(completion);
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
if let Some(usage) = &props.usage {
stage.usage = usage.usage;
stage.model = Some(usage.model().clone());
}
stage.billing_by_model.clone_from(&props.billing_by_model);
stage.usage_by_model.clone_from(&props.usage_by_model);
stage.state = StageState::from(outcome.status);
}
EventBody::StageFailed(props) => {
@ -541,11 +540,11 @@ impl RunProjectionReducer for RunProjection {
timestamp: ts,
});
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
if let Some(usage) = &props.usage {
stage.usage = usage.usage;
stage.model = Some(usage.model().clone());
}
stage.billing_by_model.clone_from(&props.billing_by_model);
stage.usage_by_model.clone_from(&props.usage_by_model);
stage.state =
stage_state_from_failure(props.will_retry, failure_category, stage.termination);
}
@ -718,7 +717,7 @@ fn apply_agent_event(
// fabro-only arms below read the same event. While the stage runs, its
// usage is that fold's: the tree's tokens, the root's and every
// subagent's, with whatever cost the provider reported. The terminal
// billing then brings the catalog's price for the same tokens.
// usage then brings the catalog's price for the same tokens.
if let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) {
let agent = stage.agent.get_or_insert_default();
agent.apply(&props.event);
@ -814,17 +813,10 @@ fn apply_agent_event(
}
/// A running stage's usage, from its agent's fold: the tree's tokens and the
/// cost the provider reported for them, `None` when it reported none.
fn live_usage(agent: &SessionProjection) -> BilledTokenCounts {
let (descendants, descendant_cost) = agent.descendant_usage();
let mut cost = agent.cost_usd_micros;
if let Some(descendant_cost) = descendant_cost {
cost = Some(cost.unwrap_or(0).saturating_add(descendant_cost));
}
BilledTokenCounts::from_token_counts(
TokenCounts::from(agent.usage.saturating_add(descendants)),
cost.map(|cost| i64::try_from(cost).unwrap_or(i64::MAX)),
)
/// cost the provider reported for them, `None` once any of them went
/// unpriced.
fn live_usage(agent: &SessionProjection) -> Usage {
agent.usage.saturating_add(agent.descendant_usage())
}
/// The model reference for a message the stage's session produced.
@ -1188,7 +1180,7 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run {
.conclusion
.as_ref()
.map(|conclusion| conclusion.timing);
let total_usd_micros = projected_billing(state).total_usd_micros;
let usage = projected_usage(state);
Run {
id: *run_id,
@ -1232,10 +1224,8 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run {
completed_at,
},
timing: run_timing,
billing: total_usd_micros.map(|total_usd_micros| RunBillingSummary {
total_usd_micros: Some(total_usd_micros),
}),
size: RunSize::from_total_usd_micros(total_usd_micros),
usage,
size: RunSize::from_cost(usage.cost),
ask_fabro: AskFabro::default(),
diff: diff_summary,
pull_request: state.pull_request.clone(),
@ -1248,22 +1238,23 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run {
}
}
pub(crate) fn projected_billing(state: &RunProjection) -> BilledTokenCounts {
if let Some(billing) = state
/// The run's usage: the conclusion's total once the run ended, else the sum
/// of every non-boundary stage's usage so far.
pub(crate) fn projected_usage(state: &RunProjection) -> Usage {
if let Some(usage) = state
.conclusion
.as_ref()
.and_then(|conclusion| conclusion.billing.as_ref())
.and_then(|conclusion| conclusion.usage)
{
return billing.clone();
return usage;
}
let mut billing = BilledTokenCounts::default();
for (stage_id, stage) in state.iter_stages() {
if !state.is_boundary_stage(stage_id.node_id()) {
billing.add_counts(&stage.usage);
}
}
billing
sum_usage(
state
.iter_stages()
.filter(|(stage_id, _)| !state.is_boundary_stage(stage_id.node_id()))
.map(|(_, stage)| stage.usage),
)
}
fn run_models(state: &RunProjection) -> Vec<RunModel> {
@ -1326,7 +1317,7 @@ fn conclusion_from_completed(
timestamp: DateTime<Utc>,
) -> Result<Conclusion> {
let (stages, total_retries) =
billing_rollup::billing_rollup_from_projection(projection).conclusion_stages(projection);
usage_rollup::usage_rollup_from_projection(projection).conclusion_stages(projection);
Ok(Conclusion {
timestamp,
status: StageOutcome::from_str(&props.status)
@ -1335,7 +1326,7 @@ fn conclusion_from_completed(
failure: None,
final_git_commit_sha: props.final_git_commit_sha.clone(),
stages,
billing: props.billing.clone(),
usage: props.usage,
total_retries,
diff: RunDiff {
patch: props.final_patch.clone(),
@ -1350,7 +1341,7 @@ fn conclusion_from_failed(
timestamp: DateTime<Utc>,
) -> Conclusion {
let (stages, total_retries) =
billing_rollup::billing_rollup_from_projection(projection).conclusion_stages(projection);
usage_rollup::usage_rollup_from_projection(projection).conclusion_stages(projection);
Conclusion {
timestamp,
status: StageOutcome::Failed {
@ -1360,7 +1351,7 @@ fn conclusion_from_failed(
failure: Some(props.failure.clone()),
final_git_commit_sha: props.final_git_commit_sha.clone(),
stages,
billing: props.billing.clone(),
usage: props.usage,
total_retries,
diff: RunDiff {
patch: props.final_patch.clone(),
@ -1432,7 +1423,7 @@ fn stage_visit(
.or_else(|| state.current_visit_for(node_id))
}
fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome<Option<BilledModelUsage>> {
fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome<Option<ModelUsage>> {
Outcome {
status: props.status,
preferred_label: props.preferred_label.clone(),
@ -1446,15 +1437,15 @@ fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome<Option<Bille
jump_to_node: props.jump_to_node.clone(),
notes: props.notes.clone(),
failure: props.failure.clone(),
usage: props.billing.clone(),
usage_by_model: props.billing_by_model.clone(),
usage: props.usage.clone(),
usage_by_model: props.usage_by_model.clone(),
files_touched: props.files_touched.clone(),
timing: Some(props.timing),
}
}
fn stage_completion_from_outcome(
outcome: &Outcome<Option<BilledModelUsage>>,
outcome: &Outcome<Option<ModelUsage>>,
timestamp: DateTime<Utc>,
) -> StageCompletion {
StageCompletion {
@ -1511,17 +1502,17 @@ mod tests {
};
use fabro_types::settings::run::DockerfileSource;
use fabro_types::{
AgentBackend, AttrValue, AutomationRef, BilledModelUsage, BilledTokenCounts, BlobHash,
BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody,
FailureCategory, FailureDetail, FailureReason, Graph, Node, Outcome, ParallelBranchId,
PendingReason, PullRequestCreationStatus, PullRequestLink, QuestionType, RunApprovalState,
RunBillingSummary, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec, RunStatus,
SandboxProviderKind, StageHandler, StageModelUsage, StageOutcome, StageState, StageTiming,
SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support,
AgentBackend, AttrValue, AutomationRef, BlobHash, BlockedReason, Checkpoint,
CheckpointRecord, CommandTermination, EventBody, FailureCategory, FailureDetail,
FailureReason, Graph, ModelUsage, Node, Outcome, ParallelBranchId, PendingReason,
PullRequestCreationStatus, PullRequestLink, QuestionType, RunApprovalState,
RunControlAction, RunDiff, RunEvent, RunSize, RunSpec, RunStatus, SandboxProviderKind,
StageHandler, StageModelUsage, StageOutcome, StageState, StageTiming, SuccessReason,
WorkflowSettings, first_event_seq, fixtures, test_support,
};
use lithos_llm::types::{ReasoningEffort, Speed, TokenCounts};
use lithos_llm::types::{Cost, CostSource, ReasoningEffort, Speed, TokenCounts, Usage};
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind, TokenUsage,
CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind,
};
use pebble_coding_agent::tools::ToolOutputMetadata;
use serde_json::json;
@ -2060,26 +2051,24 @@ mod tests {
event
}
fn test_usage(model_id: &str, input_tokens: i64, output_tokens: i64) -> BilledModelUsage {
fn test_usage(model_id: &str, input_tokens: u64, output_tokens: u64) -> ModelUsage {
serde_json::from_value(json!({
"model": { "provider": "openai", "model_id": model_id },
"tokens": {
"input": input_tokens,
"output": output_tokens
},
"total_usd_micros": input_tokens + output_tokens
"usage": {
"tokens": {
"input": input_tokens,
"output": output_tokens
},
"cost": { "usd_micros": input_tokens + output_tokens, "source": "catalog" }
}
}))
.unwrap()
}
fn usage_json(usage: &BilledModelUsage) -> serde_json::Value {
fn usage_json(usage: &ModelUsage) -> serde_json::Value {
serde_json::to_value(usage).unwrap()
}
fn usage_counts(usage: &BilledModelUsage) -> BilledTokenCounts {
BilledTokenCounts::from_billed_usage(std::slice::from_ref(usage))
}
fn test_run_spec() -> RunSpec {
RunSpec {
graph_source: Some("digraph test {}".to_string()),
@ -2627,11 +2616,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
);
@ -3318,8 +3306,8 @@ mod tests {
status: StageOutcome::Succeeded,
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: Some(usage.clone()),
usage_by_model: Vec::new(),
usage: Some(usage.clone()),
failure: None,
notes: None,
files_touched: Vec::new(),
@ -3339,7 +3327,7 @@ mod tests {
let stage = state.stage(&StageId::new("build", 1)).unwrap();
assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(789));
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
}
@ -3375,7 +3363,7 @@ mod tests {
},
"will_retry": false,
"timing": {"wall_time_ms": 654, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0},
"billing": usage_json(&usage)
"usage": usage_json(&usage)
}),
Some("build"),
))
@ -3383,7 +3371,7 @@ mod tests {
let stage = state.stage(&stage_id).unwrap();
assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(654));
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
}
@ -3406,8 +3394,8 @@ mod tests {
status: StageOutcome::Succeeded,
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: Some(usage),
usage_by_model: Vec::new(),
usage: Some(usage),
failure: None,
notes: None,
files_touched: Vec::new(),
@ -3429,10 +3417,10 @@ mod tests {
let first_stage = state.stage(&StageId::new("build", 1)).unwrap();
let second_stage = state.stage(&StageId::new("build", 2)).unwrap();
assert_eq!(first_stage.timing.map(|t| t.wall_time_ms), Some(111));
assert_eq!(first_stage.usage, usage_counts(&first_usage));
assert_eq!(first_stage.usage, first_usage.usage);
assert_eq!(first_stage.model.as_ref(), Some(first_usage.model()));
assert_eq!(second_stage.timing.map(|t| t.wall_time_ms), Some(222));
assert_eq!(second_stage.usage, usage_counts(&second_usage));
assert_eq!(second_stage.usage, second_usage.usage);
assert_eq!(second_stage.model.as_ref(), Some(second_usage.model()));
}
@ -3451,8 +3439,8 @@ mod tests {
status: StageOutcome::Succeeded,
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: Some(usage.clone()),
usage_by_model: Vec::new(),
usage: Some(usage.clone()),
failure: None,
notes: None,
files_touched: Vec::new(),
@ -3476,7 +3464,7 @@ mod tests {
);
let stage = state.stage(&scoped_stage_id).unwrap();
assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(333));
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
assert_eq!(stage.response.as_deref(), Some("done"));
}
@ -3491,15 +3479,15 @@ mod tests {
.apply_event(&test_stage_event(
3,
EventBody::StageFailed(StageFailedProps {
index: 0,
failure: Some(fabro_types::FailureDetail::new(
index: 0,
failure: Some(fabro_types::FailureDetail::new(
"try again",
fabro_types::FailureCategory::TransientInfra,
)),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(444),
billing_by_model: Vec::new(),
billing: Some(usage.clone()),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(444),
usage_by_model: Vec::new(),
usage: Some(usage.clone()),
}),
scoped_stage_id.clone(),
))
@ -3511,7 +3499,7 @@ mod tests {
);
let stage = state.stage(&scoped_stage_id).unwrap();
assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(444));
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
let completion = stage.completion.as_ref().unwrap();
assert_eq!(completion.outcome, StageOutcome::Failed {
@ -4251,39 +4239,44 @@ mod tests {
(7, "zebra", 2, 800, 200),
] {
let mut props = completed_props(millis, StageOutcome::Succeeded);
props.billing = Some(test_usage("test-model", tokens, 10));
props.usage = Some(test_usage("test-model", tokens, 10));
events.push(test_stage_event(
seq,
EventBody::StageCompleted(props),
StageId::new(node, visit),
));
}
events.push(test_raw_event(8, "checkpoint.completed", &json!({
"status": "succeeded",
"current_node": "zebra",
"completed_nodes": ["apple", "zebra", "zebra"],
"node_retries": { "zebra": 3, "apple": 1 },
"node_outcomes": {
"apple": Outcome::<Option<BilledModelUsage>>::success(),
"zebra": Outcome::<Option<BilledModelUsage>>::success(),
"skipped": Outcome::<Option<BilledModelUsage>>::skipped("condition was false")
},
"context_values": {},
"node_visits": { "zebra": 2, "apple": 1, "skipped": 1 },
"git_commit_sha": "checkpoint-sha"
}), Some("zebra")));
let terminal_billing = usage_counts(&test_usage("test-model", 320, 30));
events.push(test_raw_event(
8,
"checkpoint.completed",
&json!({
"status": "succeeded",
"current_node": "zebra",
"completed_nodes": ["apple", "zebra", "zebra"],
"node_retries": { "zebra": 3, "apple": 1 },
"node_outcomes": {
"apple": Outcome::<Option<ModelUsage>>::success(),
"zebra": Outcome::<Option<ModelUsage>>::success(),
"skipped": Outcome::<Option<ModelUsage>>::skipped("condition was false")
},
"context_values": {},
"node_visits": { "zebra": 2, "apple": 1, "skipped": 1 },
"git_commit_sha": "checkpoint-sha"
}),
Some("zebra"),
));
let terminal_usage = test_usage("test-model", 320, 30).usage;
let terminal_props = if terminal_name == "run.completed" {
json!({
"status": "succeeded", "reason": "completed",
"timing": fabro_types::RunTiming::wall_only(9000),
"artifact_count": 0, "billing": terminal_billing,
"artifact_count": 0, "usage": terminal_usage,
"final_git_commit_sha": "final-sha", "final_patch": "final patch"
})
} else {
let mut props = run_failed_props(FailureReason::WorkflowError);
props.timing = fabro_types::RunTiming::wall_only(9000);
props.billing = Some(terminal_billing.clone());
props.usage = Some(terminal_usage);
props.final_git_commit_sha = Some("final-sha".to_string());
props.final_patch = Some("final patch".to_string());
serde_json::to_value(props).unwrap()
@ -4309,7 +4302,7 @@ mod tests {
);
assert_eq!(conclusion.timestamp, events.last().unwrap().event.ts);
assert_eq!(conclusion.timing.wall_time_ms, 9000);
assert_eq!(conclusion.billing, Some(terminal_billing));
assert_eq!(conclusion.usage, Some(terminal_usage));
assert_eq!(
conclusion.final_git_commit_sha.as_deref(),
Some("final-sha")
@ -4332,7 +4325,19 @@ mod tests {
"tool_time_ms": 0,
"active_time_ms": 0
},
"billing_usd_micros": 320,
"usage": {
"tokens": {
"input": 300,
"output": 20,
"reasoning": 0,
"cache_read": 0,
"cache_write": 0
},
"cost": {
"usd_micros": 320,
"source": "catalog"
}
},
"retries": 2
},
{
@ -4344,7 +4349,19 @@ mod tests {
"tool_time_ms": 0,
"active_time_ms": 0
},
"billing_usd_micros": 30,
"usage": {
"tokens": {
"input": 20,
"output": 10,
"reasoning": 0,
"cache_read": 0,
"cache_write": 0
},
"cost": {
"usd_micros": 30,
"source": "catalog"
}
},
"retries": 0
},
{
@ -4356,6 +4373,15 @@ mod tests {
"tool_time_ms": 0,
"active_time_ms": 0
},
"usage": {
"tokens": {
"input": 0,
"output": 0,
"reasoning": 0,
"cache_read": 0,
"cache_write": 0
}
},
"retries": 0
}
],
@ -4397,7 +4423,7 @@ mod tests {
final_git_commit_sha: Some("abc123".to_string()),
final_patch: Some(patch.to_string()),
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -4543,7 +4569,7 @@ mod tests {
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -4584,7 +4610,7 @@ mod tests {
final_git_commit_sha: Some("abc123".to_string()),
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -4677,11 +4703,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -4933,11 +4958,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::PartialSuccess,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -5070,11 +5094,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}),
None,
))
@ -5112,8 +5135,8 @@ mod tests {
failure: Some(FailureDetail::new("boom", FailureCategory::TransientInfra)),
will_retry,
timing: fabro_types::StageTiming::wall_only(duration_ms),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
}
}
@ -5123,8 +5146,8 @@ mod tests {
failure: Some(FailureDetail::new("cancelled", FailureCategory::Canceled)),
will_retry,
timing: fabro_types::StageTiming::wall_only(duration_ms),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
}
}
@ -5144,7 +5167,7 @@ mod tests {
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
}
}
@ -5164,8 +5187,8 @@ mod tests {
status,
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),
@ -5181,19 +5204,21 @@ mod tests {
}
}
fn billed_usage() -> BilledModelUsage {
fn priced_usage() -> ModelUsage {
serde_json::from_value(json!({
"model": { "provider": "openai", "model_id": "gpt-test" },
"tokens": {
"input": 10,
"output": 5,
"reasoning": 2,
"cache_read": 3,
"cache_write": 4
},
"total_usd_micros": 123
"usage": {
"tokens": {
"input": 10,
"output": 5,
"reasoning": 2,
"cache_read": 3,
"cache_write": 4
},
"cost": { "usd_micros": 123, "source": "catalog" }
}
}))
.expect("billing fixture should deserialize")
.expect("usage fixture should deserialize")
}
fn agent_body(event: CodingEvent) -> EventBody {
@ -5207,14 +5232,12 @@ mod tests {
fn assistant_message(input: u64, output: u64) -> CodingEvent {
CodingEvent::AssistantMessage {
text: "assistant text".to_string(),
model: billed_usage().model().model_id.to_string(),
usage: TokenUsage {
model: priced_usage().model().model_id.to_string(),
usage: Usage::from(TokenCounts {
input,
output,
..TokenUsage::default()
},
cost_usd_micros: None,
cost_source: None,
..TokenCounts::default()
}),
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -5238,16 +5261,12 @@ mod tests {
})
}
fn live_counts(input_tokens: i64, output_tokens: i64) -> BilledTokenCounts {
BilledTokenCounts {
input_tokens,
output_tokens,
total_tokens: input_tokens + output_tokens,
reasoning_tokens: 0,
cache_read_tokens: 0,
cache_write_tokens: 0,
total_usd_micros: None,
}
fn live_counts(input: u64, output: u64) -> Usage {
Usage::from(TokenCounts {
input,
output,
..TokenCounts::default()
})
}
#[test]
@ -5273,7 +5292,7 @@ mod tests {
fn agent_message_accumulates_live_usage_on_stage_projection() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let model = billed_usage().model().clone();
let model = priced_usage().model().clone();
state
.apply_event(&test_stage_event(
@ -5323,14 +5342,14 @@ mod tests {
}
/// One usage rule: a stage's usage is its session tree's, live and at
/// completion. The terminal billing carries the tokens the fold already
/// completion. The terminal usage carries the tokens the fold already
/// showed plus the catalog's price, so completion changes the cost, not
/// the tokens, and keeps the split by model.
#[test]
fn stage_completed_keeps_the_trees_live_usage_and_prices_it() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let model = billed_usage().model().clone();
let model = priced_usage().model().clone();
state
.apply_event(&test_stage_event(
@ -5360,25 +5379,27 @@ mod tests {
stage_id.clone(),
))
.unwrap();
let live = state.stage(&stage_id).unwrap().usage.clone();
let live = state.stage(&stage_id).unwrap().usage;
assert_eq!(
live,
live_counts(107, 51),
"the subagent's tokens are the stage's too"
);
let tree = BilledModelUsage {
model: model.clone(),
tokens: TokenCounts {
let tree = ModelUsage::new(model.clone(), Usage {
tokens: TokenCounts {
input: 107,
output: 51,
..TokenCounts::default()
},
total_usd_micros: Some(321),
};
cost: Some(Cost {
usd_micros: 321,
source: CostSource::Catalog,
}),
});
let mut props = completed_props(42, StageOutcome::Succeeded);
props.billing = Some(tree.clone());
props.billing_by_model = vec![tree.clone()];
props.usage = Some(tree.clone());
props.usage_by_model = vec![tree.clone()];
state
.apply_event(&test_stage_event(
5,
@ -5389,17 +5410,19 @@ mod tests {
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.usage.token_counts(),
live.token_counts(),
stage.usage.tokens, live.tokens,
"completion keeps the tokens the fold showed"
);
assert_eq!(
stage.usage.total_usd_micros,
Some(321),
stage.usage.cost,
Some(Cost {
usd_micros: 321,
source: CostSource::Catalog,
}),
"and brings the catalog's price"
);
assert_eq!(stage.model.as_ref(), Some(&model));
assert_eq!(stage.billing_by_model, vec![tree]);
assert_eq!(stage.usage_by_model, vec![tree]);
}
#[test]
@ -5411,7 +5434,6 @@ mod tests {
text,
model,
usage,
cost_source,
tool_call_count,
context_window,
reasoning,
@ -5423,9 +5445,13 @@ mod tests {
agent_body(CodingEvent::AssistantMessage {
text,
model,
usage,
cost_usd_micros: Some(cost),
cost_source,
usage: Usage {
tokens: usage.tokens,
cost: Some(Cost {
usd_micros: cost,
source: CostSource::Provider,
}),
},
tool_call_count,
context_window,
reasoning,
@ -5462,11 +5488,16 @@ mod tests {
summary_token_estimate: 500,
tracked_file_count: 1,
reason: CompactionReason::Threshold,
usage: TokenUsage {
input: 30,
..TokenUsage::default()
usage: Usage {
tokens: TokenCounts {
input: 30,
..TokenCounts::default()
},
cost: Some(Cost {
usd_micros: 2,
source: CostSource::Provider,
}),
},
cost_usd_micros: Some(2),
}),
stage_id.clone(),
))
@ -5474,20 +5505,21 @@ mod tests {
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.usage,
BilledTokenCounts {
total_usd_micros: Some(7),
..live_counts(47, 6)
},
"the root's messages and compaction, the child's message, and the provider's cost"
stage.usage.tokens,
live_counts(47, 6).tokens,
"the root's messages and compaction, and the child's message"
);
assert_eq!(
stage.usage.cost, None,
"the child's unpriced message leaves the tree's cost unknown"
);
}
#[test]
fn stage_completed_without_billing_preserves_live_usage() {
fn stage_completed_without_usage_preserves_live_usage() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let model = billed_usage().model().clone();
let model = priced_usage().model().clone();
state
.apply_event(&test_stage_event(
@ -5537,7 +5569,7 @@ mod tests {
))
.unwrap();
let mut props = completed_props(42, StageOutcome::Succeeded);
props.billing = Some(usage);
props.usage = Some(usage);
state
.apply_event(&test_stage_event(
2,
@ -5549,18 +5581,16 @@ mod tests {
let summary = build_summary(&state, &fixtures::RUN_1);
assert_eq!(summary.size, RunSize::S);
assert_eq!(
summary.billing,
Some(RunBillingSummary {
total_usd_micros: Some(20_000_001),
})
summary.usage.cost.map(|cost| cost.usd_micros),
Some(20_000_001)
);
}
#[test]
fn stage_failed_replaces_live_usage_with_terminal_billing() {
fn stage_failed_replaces_live_usage_with_terminal_usage() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let usage = billed_usage();
let usage = priced_usage();
state
.apply_event(&test_stage_event(
@ -5577,7 +5607,7 @@ mod tests {
))
.unwrap();
let mut props = failed_props(42, false);
props.billing = Some(usage.clone());
props.usage = Some(usage.clone());
state
.apply_event(&test_stage_event(
3,
@ -5587,7 +5617,7 @@ mod tests {
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
}
@ -5619,7 +5649,7 @@ mod tests {
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert!(stage.usage.is_zero());
assert_eq!(stage.usage, Usage::default());
assert_eq!(stage.model, None);
assert_eq!(stage.state, StageState::Running);
}
@ -5628,7 +5658,7 @@ mod tests {
fn stage_completed_records_duration_usage_and_terminal_state() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let usage = billed_usage();
let usage = priced_usage();
state
.apply_event(&test_stage_event(
@ -5638,7 +5668,7 @@ mod tests {
))
.unwrap();
let mut props = completed_props(42, StageOutcome::Succeeded);
props.billing = Some(usage.clone());
props.usage = Some(usage.clone());
state
.apply_event(&test_event(
2,
@ -5649,7 +5679,7 @@ mod tests {
let stage = state.stage(&stage_id).unwrap();
assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(42));
assert_eq!(stage.usage, usage_counts(&usage));
assert_eq!(stage.usage, usage.usage);
assert_eq!(stage.model.as_ref(), Some(usage.model()));
assert_eq!(stage.state, StageState::Succeeded);
assert_eq!(stage.effective_state(), StageState::Succeeded);
@ -5735,15 +5765,15 @@ mod tests {
.apply_event(&test_event(
3,
EventBody::StageFailed(StageFailedProps {
index: 0,
failure: Some(FailureDetail::new(
index: 0,
failure: Some(FailureDetail::new(
"Script failed with exit code: 100\n\nCancelling due to test failure",
FailureCategory::Canceled,
)),
will_retry: false,
timing: fabro_types::StageTiming::wall_only(10),
billing_by_model: Vec::new(),
billing: None,
will_retry: false,
timing: fabro_types::StageTiming::wall_only(10),
usage_by_model: Vec::new(),
usage: None,
}),
Some("build"),
))
@ -6435,7 +6465,7 @@ mod tests {
assert!(open_bracket(&state).is_none());
// The close must not undo the rest of the message's work.
assert_eq!(state.stage(&stage_id()).unwrap().usage.input_tokens, 10);
assert_eq!(state.stage(&stage_id()).unwrap().usage.tokens.input, 10);
}
#[test]

View file

@ -3,8 +3,8 @@ use std::sync::LazyLock;
use chrono::{DateTime, Utc};
use fabro_types::{
BilledTokenCounts, EventEnvelope, Run, RunEvent, RunId, RunSize, RunStatusKind, RunTiming,
SessionId, StageId, timing,
EventEnvelope, Run, RunEvent, RunId, RunSize, RunStatusKind, RunTiming, SessionId, StageId,
timing,
};
use sqlx::pool::PoolConnection;
use sqlx::query::Query;
@ -12,7 +12,7 @@ use sqlx::sqlite::{SqliteArguments, SqliteConnection, SqliteRow};
use sqlx::{Connection as _, QueryBuilder, Row as _, Sqlite, SqlitePool, Transaction};
use strum::VariantArray as _;
use crate::run_state::{ProjectedRun, build_summary, projected_billing};
use crate::run_state::{ProjectedRun, build_summary, projected_usage};
use crate::{Error, EventPayload, Result, keys};
const INSERT_RUN_SQL: &str = r"
@ -1019,7 +1019,7 @@ impl PreparedRunSummary {
.unwrap_or(run.timestamps.created_at);
run.timing = entry.projection.live_run_timing(at);
}
let billing = normalize_billing_for_read_model(projected_billing(&entry.projection));
let usage = projected_usage(&entry.projection);
let workflow_name = run.workflow.display_name().map(str::to_string);
let repository_name = run
.repository
@ -1031,12 +1031,12 @@ impl PreparedRunSummary {
last_seq: entry.last_seq,
workflow_name,
repository_name,
input_tokens: billing.input_tokens,
output_tokens: billing.output_tokens,
reasoning_tokens: billing.reasoning_tokens,
cache_read_tokens: billing.cache_read_tokens,
cache_write_tokens: billing.cache_write_tokens,
total_usd_micros: billing.total_usd_micros,
input_tokens: column_count(usage.tokens.input),
output_tokens: column_count(usage.tokens.output),
reasoning_tokens: column_count(usage.tokens.reasoning),
cache_read_tokens: column_count(usage.tokens.cache_read),
cache_write_tokens: column_count(usage.tokens.cache_write),
total_usd_micros: usage.cost.map(|cost| column_count(cost.usd_micros)),
}
}
}
@ -1250,31 +1250,10 @@ async fn select_run_head(connection: &mut SqliteConnection, run_id: &RunId) -> R
.transpose()
}
/// Older provider codecs could persist a negative disjoint bucket when a
/// detail count exceeded its inclusive parent total. The SQLite summary is a
/// rebuildable, nonnegative read model, so normalize those legacy values here
/// without rewriting the authoritative run events.
fn normalize_billing_for_read_model(mut billing: BilledTokenCounts) -> BilledTokenCounts {
let input_total = billing
.input_tokens
.saturating_add(billing.cache_read_tokens)
.saturating_add(billing.cache_write_tokens)
.max(0);
billing.cache_read_tokens = billing.cache_read_tokens.clamp(0, input_total);
billing.cache_write_tokens = billing
.cache_write_tokens
.clamp(0, input_total - billing.cache_read_tokens);
billing.input_tokens = input_total - billing.cache_read_tokens - billing.cache_write_tokens;
let output_total = billing
.output_tokens
.saturating_add(billing.reasoning_tokens)
.max(0);
billing.reasoning_tokens = billing.reasoning_tokens.clamp(0, output_total);
billing.output_tokens = output_total - billing.reasoning_tokens;
billing.total_tokens = input_total.saturating_add(output_total);
billing.total_usd_micros = billing.total_usd_micros.map(|value| value.max(0));
billing
/// A usage count as the SQLite read model stores it: the columns are signed,
/// so a count past `i64::MAX` saturates rather than wrapping negative.
fn column_count(count: u64) -> i64 {
i64::try_from(count).unwrap_or(i64::MAX)
}
#[cfg(test)]
@ -1538,11 +1517,12 @@ mod tests {
use chrono::{DateTime, Utc};
use fabro_types::{
AutomationRef, BilledTokenCounts, BlockedReason, Conclusion, DiffSummary, EventEnvelope,
FailureReason, Graph, PendingReason, PullRequestCreationId, RunDiff, RunId, RunProjection,
RunSize, RunSpec, RunStatus, RunStatusKind, RunTiming, SessionId, StageId, StageOutcome,
SuccessReason, WorkflowSettings, test_support,
AutomationRef, BlockedReason, Conclusion, DiffSummary, EventEnvelope, FailureReason, Graph,
PendingReason, PullRequestCreationId, RunDiff, RunId, RunProjection, RunSize, RunSpec,
RunStatus, RunStatusKind, RunTiming, SessionId, StageId, StageOutcome, SuccessReason,
WorkflowSettings, test_support,
};
use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage};
use strum::VariantArray as _;
use tokio::time;
use ulid::Ulid;
@ -2908,7 +2888,7 @@ mod tests {
}
#[tokio::test]
async fn projection_persists_billing_diff_and_derived_size() {
async fn projection_persists_usage_diff_and_derived_size() {
let (_directory, store) = store().await;
let created_at = dt("2026-07-11T12:00:00Z");
let run_id = run_id(created_at.timestamp_millis().cast_unsigned(), 1);
@ -2930,14 +2910,18 @@ mod tests {
failure: None,
final_git_commit_sha: None,
stages: Vec::new(),
billing: Some(BilledTokenCounts {
input_tokens: 100,
output_tokens: 20,
total_tokens: 135,
reasoning_tokens: 5,
cache_read_tokens: 10,
cache_write_tokens: 0,
total_usd_micros: Some(21_000_000),
usage: Some(Usage {
tokens: TokenCounts {
input: 100,
output: 20,
reasoning: 5,
cache_read: 10,
cache_write: 0,
},
cost: Some(Cost {
usd_micros: 21_000_000,
source: CostSource::Catalog,
}),
}),
total_retries: 0,
diff: RunDiff {
@ -2997,47 +2981,6 @@ mod tests {
assert_eq!(run.size, RunSize::S);
}
#[tokio::test]
async fn projection_normalizes_legacy_overlapping_reasoning_tokens() {
let (_directory, store) = store().await;
let created_at = dt("2026-07-11T12:00:00Z");
let run_id = run_id(created_at.timestamp_millis().cast_unsigned(), 1);
let mut projection = projection(run_id, "legacy billing", created_at);
projection.conclusion = Some(Conclusion {
timestamp: created_at,
status: StageOutcome::Succeeded,
timing: RunTiming::default(),
failure: None,
final_git_commit_sha: None,
stages: Vec::new(),
billing: Some(BilledTokenCounts {
input_tokens: 53,
output_tokens: -7,
total_tokens: 112,
reasoning_tokens: 66,
..BilledTokenCounts::default()
}),
total_retries: 0,
diff: RunDiff::default(),
});
store
.upsert_projection(&entry(projection, 1))
.await
.unwrap();
let row = sqlx::query(
"SELECT input_tokens, output_tokens, reasoning_tokens FROM runs WHERE id = ?",
)
.bind(run_id.to_string())
.fetch_one(&store.pool)
.await
.unwrap();
assert_eq!(sqlx::Row::get::<i64, _>(&row, "input_tokens"), 53);
assert_eq!(sqlx::Row::get::<i64, _>(&row, "output_tokens"), 0);
assert_eq!(sqlx::Row::get::<i64, _>(&row, "reasoning_tokens"), 59);
}
#[tokio::test]
async fn reconcile_removes_rows_absent_from_authoritative_entries() {
let (_directory, store) = store().await;

View file

@ -5,10 +5,10 @@ use fabro_store::{RunProjection, SerializableProjection, StageId};
use fabro_types::graph::Graph;
use fabro_types::run::RunSpec;
use fabro_types::{
BilledModelUsage, BilledTokenCounts, Checkpoint, CheckpointRecord, InterviewQuestionRecord,
ParallelBranchResult, QuestionType, RunDiff, RunSandbox, RunSandboxInstance, RunSandboxPlan,
RunSandboxRuntime, RunStatus, SandboxProviderKind, StageCompletion, StageModelUsage,
StageOutcome, StartRecord, first_event_seq, fixtures, test_support,
Checkpoint, CheckpointRecord, InterviewQuestionRecord, ModelUsage, ParallelBranchResult,
QuestionType, RunDiff, RunSandbox, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime,
RunStatus, SandboxProviderKind, StageCompletion, StageModelUsage, StageOutcome, StartRecord,
first_event_seq, fixtures, test_support,
};
use serde_json::json;
@ -47,14 +47,16 @@ fn sample_checkpoint() -> Checkpoint {
}
}
fn sample_usage() -> BilledModelUsage {
fn sample_usage() -> ModelUsage {
serde_json::from_value(json!({
"model": { "provider": "openai", "model_id": "gpt-5.2" },
"tokens": {
"input": 123,
"output": 45
},
"total_usd_micros": 168
"usage": {
"tokens": {
"input": 123,
"output": 45
},
"cost": { "usd_micros": 168, "source": "catalog" }
}
}))
.expect("sample usage should deserialize")
}
@ -137,15 +139,14 @@ fn serializable_projection_round_trips_and_trims_bulky_node_fields() {
stage.parallel_results = Some(parallel_results.clone());
stage.timing = Some(fabro_types::StageTiming::wall_only(1234));
let usage = sample_usage();
let usage_counts = BilledTokenCounts::from_billed_usage(std::slice::from_ref(&usage));
stage.usage = usage_counts.clone();
stage.usage = usage.usage;
stage.model = Some(usage.model().clone());
stage.output = Some("output".to_string());
let serialized = serde_json::to_value(SerializableProjection(&projection))
.expect("projection should serialize");
assert_eq!(
serialized["stages"]["build@2"]["usage"]["input_tokens"],
serialized["stages"]["build@2"]["usage"]["tokens"]["input"],
json!(123)
);
assert_eq!(
@ -196,7 +197,7 @@ fn serializable_projection_round_trips_and_trims_bulky_node_fields() {
assert_eq!(node.script_timing, Some(json!({ "duration_ms": 10 })));
assert_eq!(node.parallel_results, Some(parallel_results));
assert_eq!(node.timing.map(|t| t.wall_time_ms), Some(1234));
assert_eq!(node.usage, usage_counts);
assert_eq!(node.usage, usage.usage);
assert_eq!(node.model.as_ref(), Some(usage.model()));
}

View file

@ -311,6 +311,7 @@ fn format_tool_error(err: &anyhow::Error) -> String {
#[cfg(test)]
mod tests {
use chrono::{TimeZone, Utc};
use fabro_api::types::Usage;
use fabro_types::{
RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef, test_support,
};
@ -457,7 +458,7 @@ mod tests {
completed_at: None,
},
timing: None,
billing: None,
usage: Usage::default(),
size: fabro_types::RunSize::default(),
ask_fabro: fabro_types::AskFabro::default(),
diff: None,

View file

@ -279,6 +279,7 @@ mod tests {
use std::collections::HashMap;
use chrono::{TimeZone, Utc};
use fabro_api::types::Usage;
use fabro_types::{
GitRunTarget, Run, RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps,
WorkflowRef, test_support,
@ -639,7 +640,7 @@ mod tests {
completed_at: None,
},
timing: None,
billing: None,
usage: Usage::default(),
size: fabro_types::RunSize::default(),
ask_fabro: fabro_types::AskFabro::default(),
diff: None,

View file

@ -450,6 +450,7 @@ mod tests {
use async_trait::async_trait;
use chrono::{TimeZone, Utc};
use fabro_api::types::Usage;
use fabro_types::{
EventEnvelope, FailureReason, Run, RunId, RunLifecycle, RunLinks, RunOrigin, RunProjection,
RunStatus, RunTimestamps, WorkflowRef, test_support,
@ -711,7 +712,7 @@ mod tests {
completed_at: None,
},
timing: None,
billing: None,
usage: Usage::default(),
size: fabro_types::RunSize::default(),
ask_fabro: fabro_types::AskFabro::default(),
diff: None,

View file

@ -293,6 +293,7 @@ mod tests {
use std::collections::HashMap;
use chrono::{TimeZone, Utc};
use fabro_api::types::Usage;
use fabro_types::{
RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef, test_support,
};
@ -468,7 +469,7 @@ mod tests {
completed_at: None,
},
timing: None,
billing: None,
usage: Usage::default(),
size: fabro_types::RunSize::default(),
ask_fabro: fabro_types::AskFabro::default(),
diff: None,

View file

@ -2128,15 +2128,15 @@ mod tests {
// 3. Outcome → StageFailed event
let failure = outcome.failure.clone().unwrap();
let event = Event::StageFailed {
node_id: "code".into(),
name: "code".into(),
index: 0,
failure: failure.clone(),
will_retry: false,
timing: fabro_types::StageTiming::wall_only(0),
billing_by_model: Vec::new(),
billing: None,
actor: None,
node_id: "code".into(),
name: "code".into(),
index: 0,
failure: failure.clone(),
will_retry: false,
timing: fabro_types::StageTiming::wall_only(0),
usage_by_model: Vec::new(),
usage: None,
actor: None,
};
// 4. Verify classification survived all the way through

View file

@ -235,21 +235,19 @@ fn event_body_from_event(event: &Event) -> EventBody {
artifact_count,
status,
reason,
total_usd_micros,
final_git_commit_sha,
final_patch,
diff_summary,
billing,
usage,
} => EventBody::RunCompleted(fabro_types::RunCompletedProps {
timing: *timing,
artifact_count: *artifact_count,
status: status.clone(),
reason: *reason,
total_usd_micros: *total_usd_micros,
final_git_commit_sha: final_git_commit_sha.clone(),
final_patch: final_patch.clone(),
diff_summary: *diff_summary,
billing: billing.clone(),
usage: *usage,
}),
Event::WorkflowRunFailed {
failure,
@ -257,14 +255,14 @@ fn event_body_from_event(event: &Event) -> EventBody {
final_git_commit_sha,
final_patch,
diff_summary,
billing,
usage,
} => EventBody::RunFailed(fabro_types::RunFailedProps {
failure: failure.clone(),
timing: *timing,
final_git_commit_sha: final_git_commit_sha.clone(),
final_patch: final_patch.clone(),
diff_summary: *diff_summary,
billing: billing.clone(),
usage: *usage,
}),
Event::RunNotice {
level,
@ -343,8 +341,8 @@ fn event_body_from_event(event: &Event) -> EventBody {
status,
preferred_label,
suggested_next_ids,
billing,
billing_by_model,
usage,
usage_by_model,
failure,
notes,
files_touched,
@ -364,8 +362,8 @@ fn event_body_from_event(event: &Event) -> EventBody {
status: stage_status_from_string(status),
preferred_label: preferred_label.clone(),
suggested_next_ids: suggested_next_ids.clone(),
billing: billing.clone(),
billing_by_model: billing_by_model.clone(),
usage: usage.clone(),
usage_by_model: usage_by_model.clone(),
failure: failure.clone(),
notes: notes.clone(),
files_touched: files_touched.clone(),
@ -384,16 +382,16 @@ fn event_body_from_event(event: &Event) -> EventBody {
failure,
will_retry,
timing,
billing,
billing_by_model,
usage,
usage_by_model,
..
} => EventBody::StageFailed(fabro_types::StageFailedProps {
index: *index,
failure: Some(failure.clone()),
will_retry: *will_retry,
timing: *timing,
billing: billing.clone(),
billing_by_model: billing_by_model.clone(),
usage: usage.clone(),
usage_by_model: usage_by_model.clone(),
}),
Event::StageRetrying {
index,
@ -624,13 +622,13 @@ fn event_body_from_event(event: &Event) -> EventBody {
response,
model,
provider,
billing,
usage,
..
} => EventBody::PromptCompleted(fabro_types::PromptCompletedProps {
response: response.clone(),
model: model.clone(),
provider: provider.clone(),
billing: billing.clone(),
usage: usage.clone(),
}),
Event::Agent {
stage,
@ -1021,7 +1019,9 @@ mod tests {
};
use chrono::Utc;
use lithos_llm::types::ReasoningOutput;
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, TokenUsage};
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, Cost, CostSource, TokenCounts, Usage,
};
use super::*;
use crate::error::Error;
@ -1063,8 +1063,8 @@ mod tests {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),
@ -1109,8 +1109,8 @@ mod tests {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),
@ -1135,18 +1135,18 @@ mod tests {
fn run_event_stage_failure_keeps_failure_detail() {
let usage = test_usage("gpt-5.2", 321, 54);
let stored = to_run_event(&fixtures::RUN_3, &Event::StageFailed {
node_id: "code".to_string(),
name: "Code".to_string(),
index: 1,
failure: FailureDetail::new(
node_id: "code".to_string(),
name: "Code".to_string(),
index: 1,
failure: FailureDetail::new(
"lint failed",
crate::outcome::FailureCategory::Deterministic,
),
will_retry: true,
timing: ::fabro_types::StageTiming::wall_only(5000),
billing_by_model: Vec::new(),
billing: Some(usage.clone()),
actor: None,
will_retry: true,
timing: ::fabro_types::StageTiming::wall_only(5000),
usage_by_model: Vec::new(),
usage: Some(usage.clone()),
actor: None,
});
assert_eq!(stored.event_name(), "stage.failed");
@ -1154,7 +1154,7 @@ mod tests {
assert_eq!(properties["failure"]["message"], "lint failed");
assert_eq!(properties["failure"]["category"], "deterministic");
assert_eq!(properties["will_retry"], true);
assert_eq!(properties["billing"], serde_json::to_value(&usage).unwrap());
assert_eq!(properties["usage"], serde_json::to_value(&usage).unwrap());
}
#[test]
@ -2252,9 +2252,7 @@ mod tests {
event: agent_event("ses_agent", CodingEvent::AssistantMessage {
text: "ok".to_string(),
model: "claude-sonnet".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -2278,9 +2276,13 @@ mod tests {
event: agent_event("ses_agent", CodingEvent::AssistantMessage {
text: String::new(),
model: "gpt-5.4".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: Some(125_000),
cost_source: Some(pebble_coding_agent::events::CostSource::Provider),
usage: Usage {
tokens: TokenCounts::default(),
cost: Some(Cost {
usd_micros: 125_000,
source: CostSource::Provider,
}),
},
tool_call_count: 1,
context_window: None,
reasoning: Some(ReasoningOutput::new(
@ -2293,7 +2295,8 @@ mod tests {
let value = stored.to_value().unwrap();
assert_eq!(value["event"], "agent.message");
let message = &value["properties"]["event"]["AssistantMessage"];
assert_eq!(message["cost_usd_micros"], 125_000);
assert_eq!(message["usage"]["cost"]["usd_micros"], 125_000);
assert_eq!(message["usage"]["cost"]["source"], "provider");
assert_eq!(
message["reasoning"]["summary"],
"inspect the conversion first"

View file

@ -1,20 +1,20 @@
use std::collections::BTreeMap;
use ::fabro_types::{
AutomationRef, BilledTokenCounts, BlobHash, BlockedReason, CommandTermination, DiffSummary,
FailureReason, ForkSourceRef, GitContext, PairId, PairMessageId, PairSystemMessageKind,
PairTarget, ParallelBranchId, ParallelBranchResult, PendingReason, PermissionLevel, Principal,
AutomationRef, BlobHash, BlockedReason, CommandTermination, DiffSummary, FailureReason,
ForkSourceRef, GitContext, PairId, PairMessageId, PairSystemMessageKind, PairTarget,
ParallelBranchId, ParallelBranchResult, PendingReason, PermissionLevel, Principal,
PullRequestCreationId, PullRequestLink, ReviewTarget, RunFailure, RunId, RunNoticeLevel,
RunPairEndedReason, RunPairFailedReason, RunProvenance, RunRunnableSource, RunTarget,
RunTiming, SandboxProviderKind, StageId, StageOutcome, StageTiming, SuccessReason,
WorkflowVersionId, run_event as fabro_types,
};
use lithos_llm::types::{ReasoningEffort, Speed};
use lithos_llm::types::{ReasoningEffort, Speed, Usage};
use pebble_coding_agent::events::CodingAgentEvent;
use serde::{Deserialize, Serialize};
use crate::error::{Error, run_failure_from_error};
use crate::outcome::{BilledModelUsage, FailureDetail, Outcome};
use crate::outcome::{FailureDetail, ModelUsage, Outcome};
/// Events emitted during workflow run execution for observability.
#[derive(Debug, Clone, Serialize, Deserialize)]
@ -185,15 +185,13 @@ pub enum Event {
status: String,
reason: SuccessReason,
#[serde(default, skip_serializing_if = "Option::is_none")]
total_usd_micros: Option<i64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
final_git_commit_sha: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
final_patch: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
diff_summary: Option<DiffSummary>,
#[serde(default, skip_serializing_if = "Option::is_none")]
billing: Option<BilledTokenCounts>,
usage: Option<Usage>,
},
WorkflowRunFailed {
failure: RunFailure,
@ -205,7 +203,7 @@ pub enum Event {
#[serde(default, skip_serializing_if = "Option::is_none")]
diff_summary: Option<DiffSummary>,
#[serde(default, skip_serializing_if = "Option::is_none")]
billing: Option<BilledTokenCounts>,
usage: Option<Usage>,
},
RunNotice {
level: RunNoticeLevel,
@ -269,9 +267,9 @@ pub enum Event {
status: String,
preferred_label: Option<String>,
suggested_next_ids: Vec<String>,
billing: Option<BilledModelUsage>,
usage: Option<ModelUsage>,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
billing_by_model: Vec<BilledModelUsage>,
usage_by_model: Vec<ModelUsage>,
#[serde(default, skip_serializing_if = "Option::is_none")]
failure: Option<FailureDetail>,
notes: Option<String>,
@ -294,17 +292,17 @@ pub enum Event {
max_attempts: usize,
},
StageFailed {
node_id: String,
name: String,
index: usize,
failure: FailureDetail,
will_retry: bool,
timing: StageTiming,
billing: Option<BilledModelUsage>,
node_id: String,
name: String,
index: usize,
failure: FailureDetail,
will_retry: bool,
timing: StageTiming,
usage: Option<ModelUsage>,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
billing_by_model: Vec<BilledModelUsage>,
usage_by_model: Vec<ModelUsage>,
#[serde(default, skip_serializing_if = "Option::is_none")]
actor: Option<Principal>,
actor: Option<Principal>,
},
StageRetrying {
node_id: String,
@ -495,7 +493,7 @@ pub enum Event {
model: String,
provider: String,
#[serde(default, skip_serializing_if = "Option::is_none")]
billing: Option<BilledModelUsage>,
usage: Option<ModelUsage>,
},
/// One coding-agent event, tagged with the workflow stage that produced
/// it. Pebble's envelope is kept whole: `seq`, `stream_id`, session ids,
@ -819,7 +817,7 @@ impl Event {
final_git_commit_sha: Option<String>,
final_patch: Option<String>,
diff_summary: Option<DiffSummary>,
billing: Option<BilledTokenCounts>,
usage: Option<Usage>,
) -> Self {
Self::WorkflowRunFailed {
failure: run_failure_from_error(error, reason),
@ -827,7 +825,7 @@ impl Event {
final_git_commit_sha,
final_patch,
diff_summary,
billing,
usage,
}
}

View file

@ -32,7 +32,7 @@ pub fn event_payload_from_redacted_json(line: &str, run_id: &RunId) -> Result<Ev
mod tests {
use ::fabro_types::{fixtures, run_event as fabro_types};
use lithos_llm::types::ReasoningOutput;
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, TokenUsage};
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, Usage};
use super::*;
use crate::event::{Event, to_run_event};
@ -128,9 +128,7 @@ mod tests {
CodingEvent::AssistantMessage {
text: "done".to_string(),
model: "gpt-5.4".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 0,
context_window: None,
reasoning: Some(ReasoningOutput::new(

View file

@ -355,7 +355,7 @@ mod tests {
use ::fabro_types::{Graph, RunNoticeLevel, WorkflowSettings, fixtures};
use fabro_types::test_support;
use lithos_llm::types::ReasoningOutput;
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, TokenUsage};
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, Usage};
use tokio::sync::Mutex as AsyncMutex;
use super::*;
@ -449,9 +449,7 @@ mod tests {
CodingEvent::AssistantMessage {
text: String::new(),
model: "gpt-5.4".to_string(),
usage: TokenUsage::default(),
cost_usd_micros: None,
cost_source: None,
usage: Usage::default(),
tool_call_count: 1,
context_window: None,
reasoning: Some(ReasoningOutput::new(

View file

@ -572,7 +572,7 @@ mod tests {
response: "world".into(),
model: "gpt-5.4".into(),
provider: "openai".into(),
billing: None,
usage: None,
})
.await
.unwrap();
@ -584,8 +584,8 @@ mod tests {
status: "succeeded".into(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),

View file

@ -10,7 +10,7 @@ use fabro_graphviz::graph::types::{Edge as GvEdge, Graph as GvGraph, Node as GvN
use fabro_types::ResolvedOnFailure;
use crate::context::{self, Context};
use crate::outcome::{BilledModelUsage, Outcome};
use crate::outcome::{ModelUsage, Outcome};
// ---- WorkflowNode ----
@ -78,7 +78,7 @@ impl WorkflowGraph {
impl Graph for WorkflowGraph {
type Node = WorkflowNode;
type Edge = WorkflowEdge;
type Meta = Option<BilledModelUsage>;
type Meta = Option<ModelUsage>;
fn get_node(&self, id: &str) -> Option<Self::Node> {
self.0

View file

@ -19,7 +19,7 @@ use crate::context::{Context, WorkflowContext, keys};
use crate::error::Error;
use crate::event::{Emitter, Event, StageScope};
use crate::interview_runtime::WorkflowHumanInput;
use crate::outcome::{BilledModelUsage, Outcome, OutcomeExt};
use crate::outcome::{ModelUsage, Outcome, OutcomeExt};
const LAST_FILE_ROUTING_EXTENSIONS: &[&str] = &["json", "md"];
@ -31,12 +31,12 @@ const LAST_FILE_ROUTING_EXTENSIONS: &[&str] = &["json", "md"];
pub enum CodergenResult {
Text {
text: String,
/// The stage's billing: for an agent, the whole session tree's
/// The stage's usage: for an agent, the whole session tree's
/// tokens under the root's route.
usage: Option<BilledModelUsage>,
usage: Option<ModelUsage>,
/// `usage` split by model, when the backend billed subagents at
/// their own models. Empty when `usage` is the one row.
usage_by_model: Vec<BilledModelUsage>,
usage_by_model: Vec<ModelUsage>,
files_touched: Vec<String>,
last_file_touched: Option<String>,
/// Active timing observed by the backend. The wall field is ignored by
@ -380,7 +380,7 @@ impl Handler for AgentHandler {
response: response_text.clone(),
model: response_model,
provider: response_provider,
billing: stage_usage.clone(),
usage: stage_usage.clone(),
},
&stage_scope,
);

View file

@ -24,12 +24,12 @@ use fabro_mcp::pebble::pebble_servers;
use fabro_sandbox::{RunSandbox, SecretRedactor};
use fabro_types::settings::run::RunModelControls;
use fabro_types::{
AgentProfileKind, BilledModelUsage, ModelRef, PermissionLevel, SessionCapability, StageId,
StageTiming, UsdMicros, billing,
AgentProfileKind, ModelRef, ModelUsage, PermissionLevel, SessionCapability, StageId,
StageTiming,
};
use fabro_util::home::Home;
use lithos_llm::catalog::{ModelId, ProviderId};
use lithos_llm::types::{Message as LlmMessage, Role, TokenCounts};
use lithos_llm::types::{Message as LlmMessage, Role, Usage};
use pebble_agent::ToolMiddleware;
use pebble_coding_agent::environment::Environment;
use pebble_coding_agent::events::{CodingAgentEvent, EventSink, EventSinkError};
@ -63,7 +63,7 @@ use crate::context::keys::Fidelity;
use crate::error::Error;
use crate::event::{Emitter, Event, StageScope};
use crate::model_fallback::{ModelFallbackNotice, ModelFallbackPolicy};
use crate::outcome::{Outcome, billed_model_usage_from_llm};
use crate::outcome::{Outcome, model_usage_from_llm, with_reported_cost};
use crate::services::FabroRunToolServices;
use crate::steering_hub::SteeringHub;
use crate::web_search::{self, SearchSecrets};
@ -174,7 +174,7 @@ struct WorkflowEventSink {
scope: StageScope,
/// Pebble's fold of every event this sink recorded: the stage's one
/// account of what its agent and subagents spent, wrote, and ran. The
/// store folds the same events the same way, so the stage's billing at
/// store folds the same events the same way, so the stage's usage at
/// its end is the usage the run showed live.
projection: Mutex<SessionProjection>,
}
@ -270,7 +270,6 @@ impl LiveAgent {
original_turns = compaction.original_turn_count,
preserved_turns = compaction.preserved_turn_count,
usage = ?compaction.usage,
cost_usd_micros = ?compaction.cost_usd_micros,
"agent stage compacted its conversation"
);
}
@ -320,7 +319,7 @@ impl LiveAgent {
}
}
/// The route as billing names it: provider, model, and the speed tier the
/// The route as usage names it: provider, model, and the speed tier the
/// stage asked for.
fn route_model(route: &LlmRoute) -> ModelRef {
ModelRef::new(
@ -330,62 +329,50 @@ fn route_model(route: &LlmRoute) -> ModelRef {
.with_speed(route.controls.speed)
}
/// A stage's billing from its account: the whole tree under the root's
/// A stage's usage from its account: the whole tree under the root's
/// route, and the rows that split it by model.
struct StageBilling {
total: BilledModelUsage,
by_model: Vec<BilledModelUsage>,
struct StageUsage {
total: ModelUsage,
by_model: Vec<ModelUsage>,
}
/// Bills the stage's account from the catalog: the root session at
/// Prices the stage's account from the catalog: the root session at
/// `root_model`, its route, and each descendant at its own route where the
/// catalog knows it and at the root's otherwise, so a subagent on a cheaper
/// or dearer model is priced as what it ran. A descendant on the root's
/// route joins the root's row. Where pebble carried a provider-reported
/// cost, that cost stands in for the catalog's estimate.
fn stage_billing(
/// cost, that cost stands in for the catalog's estimate. The total's cost is
/// the rows' sum, which is `None` once a row that used tokens has no cost.
fn stage_usage(
catalog: &Catalog,
root_model: &ModelRef,
account: &SessionProjection,
) -> Result<StageBilling, Error> {
let mut groups: Vec<(ModelRef, TokenCounts, Option<u64>)> = vec![(
root_model.clone(),
TokenCounts::from(account.usage),
account.cost_usd_micros,
)];
) -> Result<StageUsage, Error> {
// Each group's usage is the sum of pebble's accounts, so its cost is what
// the provider reported, or `None` once an unpriced account is in it.
let mut groups: Vec<(ModelRef, Usage)> = vec![(root_model.clone(), account.usage)];
for descendant in account.descendants.values() {
let model = descendant_model(catalog, root_model, descendant);
match groups.iter_mut().find(|(grouped, _, _)| *grouped == model) {
Some((_, tokens, cost)) => {
billing::add_usage(tokens, TokenCounts::from(descendant.usage));
add_reported_cost(cost, descendant.cost_usd_micros);
}
None => groups.push((
model,
TokenCounts::from(descendant.usage),
descendant.cost_usd_micros,
)),
match groups.iter_mut().find(|(grouped, _)| *grouped == model) {
Some((_, usage)) => *usage = usage.saturating_add(descendant.usage),
None => groups.push((model, descendant.usage)),
}
}
// The root's row first, then the others by model.
groups[1..].sort_by(|left, right| left.0.sort_key().cmp(&right.0.sort_key()));
let mut by_model = Vec::with_capacity(groups.len());
let mut total_tokens = TokenCounts::default();
let mut total_cost = None;
for (model, tokens, reported) in groups {
let row = billed_model_usage_from_llm(catalog, &model, tokens)?
.with_reported_cost(reported.map(usd_micros));
billing::add_usage(&mut total_tokens, row.tokens);
UsdMicros::accumulate(&mut total_cost, row.total_usd_micros.map(UsdMicros));
let mut total = Usage::default();
for (model, usage) in groups {
let row = with_reported_cost(
model_usage_from_llm(catalog, &model, usage.tokens)?,
usage.cost,
);
total = total.saturating_add(row.usage);
by_model.push(row);
}
Ok(StageBilling {
total: BilledModelUsage {
model: root_model.clone(),
tokens: total_tokens,
total_usd_micros: total_cost.map(|cost| cost.0),
},
Ok(StageUsage {
total: ModelUsage::new(root_model.clone(), total),
by_model,
})
}
@ -414,17 +401,6 @@ fn descendant_model(
ModelRef::new(ProviderId::new(provider), ModelId::new(model))
}
/// Folds a reported cost into a total that stays `None` until one is seen.
fn add_reported_cost(total: &mut Option<u64>, cost: Option<u64>) {
if let Some(cost) = cost {
*total = Some(total.unwrap_or(0).saturating_add(cost));
}
}
fn usd_micros(micros: u64) -> UsdMicros {
UsdMicros(i64::try_from(micros).unwrap_or(i64::MAX))
}
/// Everything one stage binds to an agent it builds or resumes.
struct StageBindings<'a> {
node_id: &'a str,
@ -780,24 +756,24 @@ impl PebbleBackend {
/// The failed outcome of an agent stage that spent before it failed: the
/// failure itself, with the session tree's usage, the files it wrote, and
/// its active time, so the run bills what the stage spent. A billing the
/// its active time, so the run records what the stage spent. A usage the
/// catalog cannot price is logged and left off.
fn failed_outcome(&self, error: &Error, live: &LiveAgent, plan: &FallbackPlan) -> Outcome {
let mut outcome = error.to_fail_outcome();
let account = live.account();
match stage_billing(
match stage_usage(
self.catalog.as_ref(),
&route_model(plan.current()),
&account,
) {
Ok(billing) => {
outcome.usage = Some(billing.total);
outcome.usage_by_model = billing.by_model;
Ok(usage) => {
outcome.usage = Some(usage.total);
outcome.usage_by_model = usage.by_model;
}
Err(billing_error) => {
Err(usage_error) => {
tracing::debug!(
error = %billing_error,
"failed agent stage could not be billed"
error = %usage_error,
"failed agent stage could not be priced"
);
}
}
@ -1001,8 +977,7 @@ impl CodergenBackend for PebbleBackend {
.map(structured_output::prompt_response_format);
let mut repair_attempts = 0_i64;
let mut previous_validation_error = None;
let mut total_usage = TokenCounts::default();
let mut total_cost = None;
let mut total_usage = Usage::default();
let mut inference_duration = Duration::ZERO;
loop {
@ -1026,11 +1001,7 @@ impl CodergenBackend for PebbleBackend {
.await;
inference_duration = inference_duration.saturating_add(inference_start.elapsed());
let completion = completion_result?;
billing::add_usage(&mut total_usage, completion.response.usage);
UsdMicros::accumulate(
&mut total_cost,
completion.response.cost.as_ref().map(UsdMicros::from_cost),
);
total_usage = total_usage.saturating_add(completion.response.usage_with_cost());
let response_text = completion.response.text();
let validation_error = if let Some(schema) = &output_schema {
@ -1057,9 +1028,12 @@ impl CodergenBackend for PebbleBackend {
continue;
}
let stage_usage =
billed_model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage)?
.with_reported_cost(total_cost);
// The provider's own cost, when every answer carried one, stands in
// for the catalog's estimate.
let stage_usage = with_reported_cost(
model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage.tokens)?,
total_usage.cost,
);
return Ok(CodergenResult::Text {
text: response_text,
@ -1260,7 +1234,7 @@ impl CodergenBackend for PebbleBackend {
};
let account = live.account();
let billing = stage_billing(
let usage = stage_usage(
self.catalog.as_ref(),
&route_model(fallback_plan.current()),
&account,
@ -1288,8 +1262,8 @@ impl CodergenBackend for PebbleBackend {
Ok(CodergenResult::Text {
text: response,
usage: Some(billing.total),
usage_by_model: billing.by_model,
usage: Some(usage.total),
usage_by_model: usage.by_model,
files_touched: account.files_touched,
last_file_touched: account.last_file_touched,
timing: StageTiming::active_only(
@ -1306,7 +1280,10 @@ mod tests {
use fabro_llm::test_support::test_catalog;
use lithos_llm::catalog::builtin;
use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, InputSource, TokenUsage};
use lithos_llm::types::TokenCounts;
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, Cost, CostSource, InputSource, Usage,
};
use super::*;
@ -1330,13 +1307,17 @@ mod tests {
CodingEvent::AssistantMessage {
text: "ok".to_string(),
model: model.to_string(),
usage: TokenUsage {
input,
output,
..TokenUsage::default()
usage: Usage {
tokens: TokenCounts {
input,
output,
..TokenCounts::default()
},
cost: cost.map(|usd_micros| Cost {
usd_micros,
source: CostSource::Provider,
}),
},
cost_usd_micros: cost,
cost_source: None,
tool_call_count: 0,
context_window: None,
reasoning: None,
@ -1354,7 +1335,7 @@ mod tests {
}
#[test]
fn stage_billing_prices_the_root_at_its_route_and_each_descendant_at_its_own() {
fn stage_usage_prices_the_root_at_its_route_and_each_descendant_at_its_own() {
let catalog = test_catalog();
let account = account(&[
root(started("openai", "gpt-5.4")),
@ -1376,47 +1357,56 @@ mod tests {
root(CodingEvent::ProcessingEnd),
]);
let billing = stage_billing(&catalog, &root_model(), &account).unwrap();
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
assert_eq!(billing.by_model.len(), 2, "{:?}", billing.by_model);
let root_row = &billing.by_model[0];
assert_eq!(usage.by_model.len(), 2, "{:?}", usage.by_model);
let root_row = &usage.by_model[0];
assert_eq!(root_row.model, root_model());
assert_eq!(
root_row.tokens.input, 111_000,
root_row.usage.tokens.input, 111_000,
"the root, the same-route child, and the unknown-route child"
);
assert_eq!(root_row.tokens.output, 26_100);
assert_eq!(root_row.usage.tokens.output, 26_100);
let root_priced =
billed_model_usage_from_llm(&catalog, &root_model(), root_row.tokens).unwrap();
assert_eq!(root_row.total_usd_micros, root_priced.total_usd_micros);
model_usage_from_llm(&catalog, &root_model(), root_row.usage.tokens).unwrap();
assert_eq!(root_row.usage.cost, root_priced.usage.cost);
assert_eq!(
root_row.usage.cost.map(|cost| cost.source),
Some(CostSource::Catalog)
);
let other_model = ModelRef::new(
ProviderId::new("anthropic"),
ModelId::new("claude-sonnet-5"),
);
let other_row = &billing.by_model[1];
let other_row = &usage.by_model[1];
assert_eq!(other_row.model, other_model);
assert_eq!(other_row.tokens.input, 20_000);
assert_eq!(other_row.tokens.output, 2_000);
assert_eq!(other_row.usage.tokens.input, 20_000);
assert_eq!(other_row.usage.tokens.output, 2_000);
let other_priced =
billed_model_usage_from_llm(&catalog, &other_model, other_row.tokens).unwrap();
assert_eq!(other_row.total_usd_micros, other_priced.total_usd_micros);
model_usage_from_llm(&catalog, &other_model, other_row.usage.tokens).unwrap();
assert_eq!(other_row.usage.cost, other_priced.usage.cost);
assert_ne!(
other_row.total_usd_micros,
billed_model_usage_from_llm(&catalog, &root_model(), other_row.tokens)
other_row.usage.cost,
model_usage_from_llm(&catalog, &root_model(), other_row.usage.tokens)
.unwrap()
.total_usd_micros,
.usage
.cost,
"priced at its own rate, not the root's"
);
// The total is the tree's tokens under the root's route, at the rows' summed
// cost.
assert_eq!(billing.total.model, root_model());
assert_eq!(billing.total.tokens.input, 131_000);
assert_eq!(billing.total.tokens.output, 28_100);
// cost, from the catalog like every row.
assert_eq!(usage.total.model, root_model());
assert_eq!(usage.total.usage.tokens.input, 131_000);
assert_eq!(usage.total.usage.tokens.output, 28_100);
assert_eq!(
billing.total.total_usd_micros,
Some(root_priced.total_usd_micros.unwrap() + other_priced.total_usd_micros.unwrap())
usage.total.usage.cost,
Some(Cost {
usd_micros: root_priced.usage.cost.unwrap().usd_micros
+ other_priced.usage.cost.unwrap().usd_micros,
source: CostSource::Catalog,
})
);
}
@ -1430,22 +1420,29 @@ mod tests {
child("ses_child", message("claude-sonnet-5", 500, 50, None)),
]);
let billing = stage_billing(&catalog, &root_model(), &account).unwrap();
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
assert_eq!(billing.by_model[0].total_usd_micros, Some(4_321));
let child_priced = billed_model_usage_from_llm(
assert_eq!(
usage.by_model[0].usage.cost,
Some(Cost {
usd_micros: 4_321,
source: CostSource::Provider,
})
);
let child_priced = model_usage_from_llm(
&catalog,
&billing.by_model[1].model,
billing.by_model[1].tokens,
&usage.by_model[1].model,
usage.by_model[1].usage.tokens,
)
.unwrap();
assert_eq!(usage.by_model[1].usage.cost, child_priced.usage.cost);
// One row reported, one priced: the sum is the application's.
assert_eq!(
billing.by_model[1].total_usd_micros,
child_priced.total_usd_micros
);
assert_eq!(
billing.total.total_usd_micros,
Some(4_321 + child_priced.total_usd_micros.unwrap())
usage.total.usage.cost,
Some(Cost {
usd_micros: 4_321 + child_priced.usage.cost.unwrap().usd_micros,
source: CostSource::Application,
})
);
}
@ -1459,14 +1456,14 @@ mod tests {
message("gpt-5.4-mini", 1_000, 100, None),
));
let billing = stage_billing(&catalog, &root_model(), &account).unwrap();
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
let child_row = billing
let child_row = usage
.by_model
.iter()
.find(|row| row.model.model_id.as_str() == "gpt-5.4-mini")
.expect("the child is billed as its answers' model on the root's provider");
assert_eq!(child_row.model.provider, root_model().provider);
assert_eq!(child_row.tokens.input, 1_000);
assert_eq!(child_row.usage.tokens.input, 1_000);
}
}

View file

@ -594,10 +594,10 @@ mod tests {
use lithos_llm::types::TokenCounts;
use super::*;
use crate::outcome::{BilledModelUsage, billed_model_usage_from_llm};
use crate::outcome::{ModelUsage, model_usage_from_llm};
fn stage_usage(model: &str, input: u64, output: u64) -> BilledModelUsage {
billed_model_usage_from_llm(
fn stage_usage(model: &str, input: u64, output: u64) -> ModelUsage {
model_usage_from_llm(
&fabro_llm::test_support::test_catalog(),
&ModelRef::new(builtin::anthropic(), ModelId::new(model)),
TokenCounts {

View file

@ -150,7 +150,7 @@ impl Handler for PromptHandler {
response: response_text.clone(),
model: response_model,
provider: response_provider,
billing: stage_usage.clone(),
usage: stage_usage.clone(),
},
&stage_scope,
);

View file

@ -61,7 +61,7 @@ pub fn extract_stage_timings_by_stage_id(
timings
}
/// Sum of timing in each node across every visit. Use for billing/usage
/// Sum of timing in each node across every visit. Use for usage
/// where a retried node should count its full time. `wall_time_ms`,
/// `inference_time_ms`, `tool_time_ms`, and `active_time_ms` are all summed
/// per node.
@ -125,8 +125,8 @@ mod duration_tests {
status: StageOutcome::Succeeded,
preferred_label: None,
suggested_next_ids: vec![],
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: vec![],
@ -159,12 +159,12 @@ mod duration_tests {
tool_call_id: None,
actor: None,
body: EventBody::StageFailed(StageFailedProps {
index: 0,
failure: None,
will_retry: true,
timing: StageTiming::wall_only(wall_time_ms),
billing_by_model: Vec::new(),
billing: None,
index: 0,
failure: None,
will_retry: true,
timing: StageTiming::wall_only(wall_time_ms),
usage_by_model: Vec::new(),
usage: None,
}),
};
EventEnvelope { seq, event }
@ -252,8 +252,8 @@ mod duration_tests {
status: StageOutcome::Succeeded,
preferred_label: None,
suggested_next_ids: vec![],
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: vec![],
@ -289,7 +289,6 @@ pub mod agent_memory;
pub mod artifact;
pub mod artifact_snapshot;
pub mod artifact_upload;
pub mod billing_rollup;
pub mod command_log;
pub(crate) mod condition;
pub mod context;
@ -319,14 +318,15 @@ mod retry;
pub mod run_control;
pub(crate) mod run_dir;
pub mod run_lookup;
pub mod usage_rollup;
pub use billing_rollup::{
ProjectionBillingByModel, ProjectionBillingRollup, ProjectionBillingStage,
billing_rollup_from_projection,
};
pub use error::{Error, FailureCategory, FailureSignature, FailureSignatureExt, Result};
pub use fabro_types::ManifestPath;
pub use steering_hub::{PairControlError, SteeringHub};
pub use usage_rollup::{
ProjectionUsageByModel, ProjectionUsageRollup, ProjectionUsageStage,
usage_rollup_from_projection,
};
pub mod run_materialization;
pub mod run_options;
pub mod run_status;

View file

@ -23,12 +23,12 @@ use crate::artifact_upload::ArtifactSink;
use crate::event::{Emitter, Event, RunNoticeCode, RunNoticeLevel};
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::lifecycle::event::stage_scope_for;
use crate::outcome::BilledModelUsage;
use crate::outcome::ModelUsage;
use crate::runtime_store::RunStoreHandle;
use crate::stage_execution::StageExecutionTracker;
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
type ArtifactIdentity = (String, String);
const ARTIFACT_UPLOAD_RETRY_DELAYS: [Duration; 3] = [

View file

@ -9,10 +9,10 @@ use fabro_core::state::ExecutionState;
use crate::error::{FailureCategory, FailureSignature, FailureSignatureExt};
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::outcome::{BilledModelUsage, OutcomeExt};
use crate::outcome::{ModelUsage, OutcomeExt};
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
/// Sub-lifecycle responsible for tracking failure signatures and tripping the
/// circuit breaker when deterministic failure cycles are detected.

View file

@ -17,12 +17,12 @@ use super::git::GitCheckpointResult;
use crate::context::{Context, WorkflowContext};
use crate::event::{Emitter, Event, StageScope};
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::outcome::{BilledModelUsage, FailureCategory, FailureDetail, Outcome, StageOutcome};
use crate::outcome::{FailureCategory, FailureDetail, ModelUsage, Outcome, StageOutcome};
use crate::stage_execution::{StageExecution, StageExecutionTracker};
use crate::{artifact, context};
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
type FailureSignatureSnapshot = (
Option<BTreeMap<String, usize>>,
Option<BTreeMap<String, usize>>,
@ -215,8 +215,8 @@ impl RunLifecycle<WorkflowGraph> for EventLifecycle {
status: StageOutcome::Succeeded.to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),
@ -241,7 +241,7 @@ impl RunLifecycle<WorkflowGraph> for EventLifecycle {
&self,
ctx: &AttemptContext<'_, WorkflowGraph>,
state: &WfRunState,
) -> CoreResult<NodeDecision<Option<BilledModelUsage>>> {
) -> CoreResult<NodeDecision<Option<ModelUsage>>> {
let gv = ctx.node.inner();
let execution = self.stage_executions.active(&gv.id);
let scope = stage_scope_from_execution(execution.as_deref(), state, &gv.id);
@ -290,8 +290,8 @@ impl RunLifecycle<WorkflowGraph> for EventLifecycle {
failure,
will_retry: true,
timing,
billing: outcome.usage.clone(),
billing_by_model: outcome.usage_by_model.clone(),
usage: outcome.usage.clone(),
usage_by_model: outcome.usage_by_model.clone(),
actor,
},
&scope,
@ -343,8 +343,8 @@ impl RunLifecycle<WorkflowGraph> for EventLifecycle {
failure,
will_retry: false,
timing,
billing: outcome.usage.clone(),
billing_by_model: outcome.usage_by_model.clone(),
usage: outcome.usage.clone(),
usage_by_model: outcome.usage_by_model.clone(),
actor,
},
&scope,
@ -359,8 +359,8 @@ impl RunLifecycle<WorkflowGraph> for EventLifecycle {
status: outcome.status.to_string(),
preferred_label: outcome.preferred_label.clone(),
suggested_next_ids: outcome.suggested_next_ids.clone(),
billing: outcome.usage.clone(),
billing_by_model: outcome.usage_by_model.clone(),
usage: outcome.usage.clone(),
usage_by_model: outcome.usage_by_model.clone(),
failure: outcome.failure.clone(),
notes: outcome.notes.clone(),
files_touched: outcome.files_touched.clone(),

View file

@ -14,11 +14,11 @@ use crate::artifact;
use crate::context::{Context, ParallelBranchPreamble, keys};
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::handler::llm::preamble::build_preamble;
use crate::outcome::{BilledModelUsage, Outcome};
use crate::outcome::{ModelUsage, Outcome};
use crate::runtime_store::RunStoreHandle;
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeDecision = NodeDecision<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeDecision = NodeDecision<Option<ModelUsage>>;
/// Graphviz edge captured from edge selection, passed to the next node's
/// before_node for fidelity/thread resolution.

View file

@ -11,7 +11,7 @@ use fabro_types::{DiffSummary, RunId};
use crate::event::{Emitter, Event, RunNoticeCode, RunNoticeLevel};
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::lifecycle::event::stage_scope_for;
use crate::outcome::BilledModelUsage;
use crate::outcome::ModelUsage;
use crate::run_options::RunOptions;
use crate::sandbox_git::{
checked_git_checkpoint, git_diff, list_diff_numstat, summarize_diff_numstat,
@ -19,8 +19,8 @@ use crate::sandbox_git::{
use crate::sandbox_git_runtime::SandboxGitRuntime;
use crate::stage_execution::StageExecutionTracker;
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
/// Result of a git checkpoint operation, shared with EventLifecycle.
#[derive(Debug, Clone)]

View file

@ -13,11 +13,11 @@ use fabro_types::RunId;
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::hook_context::set_hook_node;
use crate::outcome::{BilledModelUsage, Outcome, OutcomeExt, StageOutcome};
use crate::outcome::{ModelUsage, Outcome, OutcomeExt, StageOutcome};
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfNodeDecision = NodeDecision<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
type WfNodeDecision = NodeDecision<Option<ModelUsage>>;
/// Sub-lifecycle responsible for running workflow hooks.
pub(crate) struct HookLifecycle {

View file

@ -35,7 +35,7 @@ use crate::context;
use crate::error::FailureSignature;
use crate::event::Emitter;
use crate::graph::{WorkflowGraph, WorkflowNode};
use crate::outcome::{BilledModelUsage, Outcome};
use crate::outcome::{ModelUsage, Outcome};
use crate::run_control::RunControlState;
use crate::run_options::RunOptions;
use crate::runtime_store::RunStoreHandle;
@ -43,9 +43,9 @@ use crate::sandbox_git_runtime::SandboxGitRuntime;
use crate::services::RunLocations;
use crate::stage_execution::StageExecutionTracker;
type WfRunState = ExecutionState<Option<BilledModelUsage>>;
type WfNodeResult = NodeResult<Option<BilledModelUsage>>;
type WfNodeDecision = NodeDecision<Option<BilledModelUsage>>;
type WfRunState = ExecutionState<Option<ModelUsage>>;
type WfNodeResult = NodeResult<Option<ModelUsage>>;
type WfNodeDecision = NodeDecision<Option<ModelUsage>>;
/// Orchestrates all sub-lifecycles with explicit per-callback ordering.
/// Implements `RunLifecycle<WorkflowGraph>` by delegating to focused structs.

View file

@ -167,11 +167,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})
.await
.unwrap();

View file

@ -427,8 +427,8 @@ mod tests {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing_by_model: Vec::new(),
billing: None,
usage_by_model: Vec::new(),
usage: None,
failure: None,
notes: None,
files_touched: Vec::new(),

View file

@ -248,11 +248,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: fabro_types::SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})
.await
.unwrap();

View file

@ -297,7 +297,7 @@ pub(super) async fn execute_persisted_run(
}
/// Build a conclusion from the store and emit `run.failed` carrying the
/// rolled-up timing and billing. Shared by the engine-failure terminal path,
/// rolled-up timing and usage. Shared by the engine-failure terminal path,
/// the bootstrap/completion drop guards, and `persist_detached_failure`.
async fn emit_workflow_run_failed(
run_id: RunId,
@ -325,7 +325,7 @@ async fn emit_workflow_run_failed(
None,
None,
None,
conclusion.billing,
conclusion.usage,
);
if let Err(err) = append_event_to_sink(event_sink, &run_id, &failure_event).await {
let rendered_error = collect_chain(&err).join(": ");
@ -1303,11 +1303,12 @@ mod tests {
RunPrepareSettings,
};
use fabro_types::{
BilledModelUsage, GitContext, ManifestPath, RunTarget, StageTiming, WorkflowSettings,
fixtures, test_support,
GitContext, ManifestPath, ModelUsage, RunTarget, StageTiming, WorkflowSettings, fixtures,
test_support,
};
use fabro_vault::SecretType;
use lithos_llm::catalog::builtin;
use lithos_llm::types::Usage;
use object_store::memory::InMemory;
use super::*;
@ -2556,7 +2557,7 @@ mod tests {
run_store: &fabro_store::RunDatabase,
node_id: &str,
timing: fabro_types::StageTiming,
billing: Option<BilledModelUsage>,
usage: Option<ModelUsage>,
) {
crate::event::append_event(run_store, &fixtures::RUN_1, &Event::StageCompleted {
node_id: node_id.to_string(),
@ -2566,8 +2567,8 @@ mod tests {
status: StageOutcome::Succeeded.to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing,
billing_by_model: Vec::new(),
usage,
usage_by_model: Vec::new(),
failure: None,
notes: None,
files_touched: Vec::new(),
@ -2707,7 +2708,7 @@ mod tests {
}
#[tokio::test]
async fn persist_terminal_engine_failure_uses_conclusion_timing_and_billing() {
async fn persist_terminal_engine_failure_uses_conclusion_timing_and_usage() {
let temp = tempfile::tempdir().unwrap();
let (storage_root, run_dir) = storage_root_and_run_dir(&temp);
let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await;
@ -2748,17 +2749,11 @@ mod tests {
assert_eq!(conclusion.timing.inference_time_ms, 225);
assert_eq!(conclusion.timing.tool_time_ms, 375);
assert_eq!(conclusion.timing.active_time_ms, 600);
assert_eq!(
conclusion
.billing
.as_ref()
.map(|billing| billing.total_tokens),
Some(150),
);
assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(150),);
}
#[tokio::test]
async fn bootstrap_guard_failure_uses_conclusion_timing_and_billing() {
async fn bootstrap_guard_failure_uses_conclusion_timing_and_usage() {
let temp = tempfile::tempdir().unwrap();
let (storage_root, _run_dir) = storage_root_and_run_dir(&temp);
let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await;
@ -2787,17 +2782,11 @@ mod tests {
assert_eq!(conclusion.timing.inference_time_ms, 120);
assert_eq!(conclusion.timing.tool_time_ms, 80);
assert_eq!(conclusion.timing.active_time_ms, 200);
assert_eq!(
conclusion
.billing
.as_ref()
.map(|billing| billing.total_tokens),
Some(50),
);
assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(50),);
}
#[tokio::test]
async fn completion_guard_failure_uses_conclusion_timing_and_billing() {
async fn completion_guard_failure_uses_conclusion_timing_and_usage() {
let temp = tempfile::tempdir().unwrap();
let (storage_root, _run_dir) = storage_root_and_run_dir(&temp);
let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await;
@ -2826,13 +2815,7 @@ mod tests {
assert_eq!(conclusion.timing.inference_time_ms, 70);
assert_eq!(conclusion.timing.tool_time_ms, 30);
assert_eq!(conclusion.timing.active_time_ms, 100);
assert_eq!(
conclusion
.billing
.as_ref()
.map(|billing| billing.total_tokens),
Some(25),
);
assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(25),);
}
#[tokio::test]
@ -3165,7 +3148,7 @@ mod tests {
failure: None,
final_git_commit_sha: None,
stages: vec![],
billing: None,
usage: None,
total_retries: 0,
diff: fabro_types::RunDiff::default(),
};
@ -3215,11 +3198,10 @@ mod tests {
artifact_count: 0,
status: "succeeded".to_string(),
reason: crate::run_status::SuccessReason::Completed,
total_usd_micros: None,
final_git_commit_sha: None,
final_patch: None,
diff_summary: None,
billing: None,
usage: None,
})
.await
.unwrap();

View file

@ -2,36 +2,43 @@ pub use fabro_core::outcome::{
FailureCategory, FailureDetail, OutcomeMeta, StageOutcome, StageState,
};
use fabro_llm::lithos_catalog::Catalog;
pub use fabro_types::BilledModelUsage;
use fabro_types::{BilledTokenCounts, ModelRef};
use lithos_llm::types::TokenCounts;
use fabro_types::ModelRef;
pub use fabro_types::ModelUsage;
use lithos_llm::types::{Cost, TokenCounts, Usage};
use crate::error::{Error, FailureSignature, classify_failure_reason};
pub type Outcome = fabro_core::Outcome<Option<BilledModelUsage>>;
pub type Outcome = fabro_core::Outcome<Option<ModelUsage>>;
/// Bills `usage` on `model` from catalog pricing.
/// Prices `tokens` on `model` from the catalog: the usage carries a
/// [`CostSource::Catalog`](lithos_llm::types::CostSource::Catalog) cost when
/// the catalog has rates for the model, and no cost otherwise.
///
/// The provider must be one the catalog knows; a passthrough model on a known
/// provider is billed with no cost, since the catalog has no rates for it.
pub fn billed_model_usage_from_llm(
/// provider is priced with no cost, since the catalog has no rates for it.
pub fn model_usage_from_llm(
catalog: &Catalog,
model: &ModelRef,
usage: TokenCounts,
) -> Result<BilledModelUsage, Error> {
tokens: TokenCounts,
) -> Result<ModelUsage, Error> {
if catalog.enabled_provider(model.provider.as_str()).is_none() {
return Err(Error::Precondition(format!(
"Provider \"{}\" is not configured",
model.provider
)));
}
let cost = catalog.estimate_cost(&model.handle(), usage, model.speed);
Ok(BilledModelUsage::new(model.clone(), usage, cost))
let cost = catalog.estimate_cost(&model.handle(), tokens, model.speed);
Ok(ModelUsage::new(model.clone(), Usage { tokens, cost }))
}
/// `usage` with `cost` in place of whatever it carried, when a provider
/// reported one; `None` keeps the usage as it is.
#[must_use]
pub fn billed_token_counts_from_llm(usage: TokenCounts) -> BilledTokenCounts {
BilledTokenCounts::from_token_counts(usage, None)
pub fn with_reported_cost(mut usage: ModelUsage, cost: Option<Cost>) -> ModelUsage {
if let Some(cost) = cost {
usage.usage.cost = Some(cost);
}
usage
}
pub trait OutcomeExt: Sized {
@ -126,11 +133,11 @@ pub fn format_cost(cost: f64) -> String {
mod tests {
use fabro_llm::lithos_catalog::Catalog;
use fabro_llm::test_support::{test_catalog, test_catalog_with_overlay};
use fabro_types::{ModelRef, UsdMicros};
use fabro_types::ModelRef;
use lithos_llm::catalog::{ModelId, ProviderId, builtin};
use lithos_llm::types::{Speed, TokenCounts};
use lithos_llm::types::{Cost, CostSource, Speed, TokenCounts};
use super::{OutcomeExt, billed_model_usage_from_llm};
use super::{OutcomeExt, model_usage_from_llm, with_reported_cost};
fn model_ref(provider: ProviderId, model_id: &str, speed: Option<Speed>) -> ModelRef {
ModelRef::new(provider, ModelId::new(model_id)).with_speed(speed)
@ -141,7 +148,7 @@ mod tests {
}
#[test]
fn billed_model_usage_from_llm_bills_openai_cached_input_and_reasoning_output() {
fn model_usage_from_llm_prices_openai_cached_input_and_reasoning_output() {
// Stay under the 272k long-context tier so the standard rates apply.
let usage = TokenCounts {
input: 100_000,
@ -150,7 +157,7 @@ mod tests {
cache_read: 50_000,
..TokenCounts::default()
};
let billed = billed_model_usage_from_llm(
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "gpt-5.4", None),
usage,
@ -158,9 +165,15 @@ mod tests {
.unwrap();
// 100k input at $2.50/M + 50k cached at $0.25/M + 30k output at $15/M.
assert_eq!(billed.total_usd_micros, Some(712_500));
assert_eq!(billed.tokens().output, 25_000);
assert_eq!(billed.tokens().reasoning, 5_000);
assert_eq!(
billed.usage.cost,
Some(Cost {
usd_micros: 712_500,
source: CostSource::Catalog,
})
);
assert_eq!(billed.usage.tokens.output, 25_000);
assert_eq!(billed.usage.tokens.reasoning, 5_000);
}
#[test]
@ -170,15 +183,22 @@ mod tests {
output: 7,
..TokenCounts::default()
};
let billed = billed_model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "gpt-5.4", None),
usage,
)
.unwrap()
.with_reported_cost(Some(UsdMicros(125_000)));
let reported = Cost {
usd_micros: 125_000,
source: CostSource::Provider,
};
let billed = with_reported_cost(
model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "gpt-5.4", None),
usage,
)
.unwrap(),
Some(reported),
);
assert_eq!(billed.total_usd_micros, Some(125_000));
assert_eq!(billed.usage.cost, Some(reported));
assert_eq!(billed.usage.tokens, usage);
}
#[test]
@ -192,7 +212,7 @@ mod tests {
}
#[test]
fn billed_model_usage_from_llm_bills_anthropic_fast_mode_cache_write_pricing() {
fn model_usage_from_llm_prices_anthropic_fast_mode_cache_write_rates() {
let usage = TokenCounts {
input: 100_000,
output: 10_000,
@ -200,7 +220,7 @@ mod tests {
cache_read: 20_000,
cache_write: 30_000,
};
let billed = billed_model_usage_from_llm(
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::anthropic(), "claude-opus-5", Some(Speed::Fast)),
usage,
@ -209,11 +229,14 @@ mod tests {
// Fast rates: $10/M input, $50/M output (incl. reasoning), $1/M cache
// read, $12.50/M cache write.
assert_eq!(billed.total_usd_micros, Some(2_145_000));
assert_eq!(
billed.usage.cost.map(|cost| cost.usd_micros),
Some(2_145_000)
);
}
#[test]
fn billed_model_usage_from_llm_uses_injected_custom_catalog() {
fn model_usage_from_llm_uses_injected_custom_catalog() {
let catalog = test_catalog_with_overlay(
r#"
[providers.proxy]
@ -237,20 +260,23 @@ pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_millio
output: 500_000,
..TokenCounts::default()
};
let billed = billed_model_usage_from_llm(
let billed = model_usage_from_llm(
&catalog,
&model_ref(ProviderId::new("proxy"), "canonical-model", None),
usage,
)
.unwrap();
assert_eq!(billed.total_usd_micros, Some(2_000_000));
assert_eq!(
billed.usage.cost.map(|cost| cost.usd_micros),
Some(2_000_000)
);
assert_eq!(billed.model_id(), "canonical-model");
}
#[test]
fn passthrough_model_on_known_provider_has_no_cost() {
let billed = billed_model_usage_from_llm(
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "brand-new-model", None),
TokenCounts {
@ -260,13 +286,13 @@ pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_millio
},
)
.unwrap();
assert_eq!(billed.total_usd_micros, None);
assert_eq!(billed.tokens().input, 10);
assert_eq!(billed.usage.cost, None);
assert_eq!(billed.usage.tokens.input, 10);
}
#[test]
fn unknown_provider_is_a_precondition_failure() {
let error = billed_model_usage_from_llm(
let error = model_usage_from_llm(
&catalog(),
&model_ref(ProviderId::new("nowhere"), "model", None),
TokenCounts::default(),

Some files were not shown because too many files have changed in this diff Show more