feat(web): show stage tokens and cost in the model popover

The model indicator on a stage page hovered to provider, model, and
reasoning effort only. Seeing what a stage actually spent meant leaving
for the Billing tab, which reports per node rather than per visit.

The stage list had no token data to show, so add a per-visit `billing`
block to `GET /runs/{id}/stages`. The Billing tab's pricing rule (a
provider-reported cost wins, otherwise the server catalog prices the
tokens) was private to `billing_rollup`; move it to
`StageProjection::billed_usage` and drive both call sites from it so the
two views cannot drift.

The popover's buckets use the Billing tab's labels verbatim. It stays
scoped to one visit, so a looped node's row on the Billing tab is the sum
of what each of its visits shows here.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Bryan Helmkamp 2026-07-27 16:06:19 -04:00
parent 6efba896f4
commit 7841a77f2c
No known key found for this signature in database
11 changed files with 423 additions and 34 deletions

View file

@ -17,6 +17,7 @@ function makeStage(nodeId: string, visit: number, status: StageState): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: null,
};
}
@ -38,6 +39,15 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
},
billing: {
input_tokens: 28_640,
output_tokens: 7_550,
total_tokens: 43_690,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_usd_micros: 720_000,
},
},
{
id: "apply-changes@2",
@ -46,6 +56,14 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "apply",
visit: 2,
billing: {
input_tokens: 0,
output_tokens: 0,
total_tokens: 0,
reasoning_tokens: 0,
cache_read_tokens: 0,
cache_write_tokens: 0,
},
},
],
meta: { has_more: false },
@ -64,6 +82,10 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
});
// Each visit keeps its own tokens and cost, so the stage popover never
// shows a sibling visit's usage.
expect(result[0].billing?.total_usd_micros).toBe(720_000);
expect(result[1].billing?.total_usd_micros).toBeUndefined();
expect(formatStageLabel(result[0])).toBe("Apply Changes");
expect(result[1].id).toBe("apply-changes@2");

View file

@ -1,5 +1,6 @@
import { StageState } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
PaginatedRunStageList,
StageHandler,
StageModelUsage,
@ -27,6 +28,11 @@ export interface Stage {
resumedFromStageId: string | null;
startedAt: string | null;
providerUsed: StageModelUsage | null;
/**
* Tokens and cost for this visit alone, priced the same way the Billing tab
* prices its per-node rows. All-zero counts mean the stage called no model.
*/
billing: BilledTokenCounts | null;
}
export const ACTIVE_STAGE_STATES: ReadonlySet<StageState> = new Set([
@ -102,6 +108,7 @@ export function mapRunStagesToSidebarStages(
: "--",
startedAt: stage.started_at ?? null,
providerUsed: stage.provider_used ?? null,
billing: stage.billing ?? null,
});
}
return stages;

View file

@ -1,9 +1,13 @@
import { describe, expect, test } from "bun:test";
import { renderToStaticMarkup } from "react-dom/server";
import type { ReasoningOutput } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
ReasoningOutput,
StageModelUsage,
} from "@qltysh/fabro-api-client";
import { EventDetails } from "./run-stages";
import { EventDetails, ModelUsagePopover } from "./run-stages";
const RUN_START = "2026-04-09T12:00:00Z";
@ -72,3 +76,82 @@ describe("EventDetails reasoning", () => {
expect(html).toContain(`${"x".repeat(280)}…`);
});
});
const PROVIDER_USED: StageModelUsage = {
mode: "agent",
provider: "moonshot",
model: "kimi-k3",
reasoning_effort: "max",
};
function billing(partial: Partial<BilledTokenCounts>): BilledTokenCounts {
return {
input_tokens: 0,
output_tokens: 0,
total_tokens: 0,
reasoning_tokens: 0,
cache_read_tokens: 0,
cache_write_tokens: 0,
...partial,
};
}
function popoverMarkup(counts: BilledTokenCounts | null): string {
return renderToStaticMarkup(
<ModelUsagePopover providerUsed={PROVIDER_USED} billing={counts} />,
);
}
describe("ModelUsagePopover billing", () => {
test("shows the visit's token buckets and cost next to the model", () => {
const html = popoverMarkup(
billing({
input_tokens: 28_640,
output_tokens: 7_550,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_tokens: 43_690,
total_usd_micros: 720_000,
}),
);
expect(html).toContain("kimi-k3");
expect(html).toContain("Cache read");
expect(html).toContain("4.8k");
expect(html).toContain("Cache creation");
expect(html).toContain("1.5k");
expect(html).toContain("Uncached");
expect(html).toContain("28.6k");
// Output folds in reasoning tokens, matching the Billing tab.
expect(html).toContain("Output");
expect(html).toContain("8.8k");
expect(html).toContain("Cost");
expect(html).toContain("$0.72");
});
test("omits the token section for a stage that called no model", () => {
const html = popoverMarkup(billing({}));
expect(html).toContain("kimi-k3");
expect(html).not.toContain("Tokens");
expect(html).not.toContain("Cost");
});
test("still shows tokens when nothing priced the stage", () => {
const html = popoverMarkup(
billing({ input_tokens: 1_000, output_tokens: 500, total_tokens: 1_500 }),
);
expect(html).toContain("Uncached");
expect(html).toContain("1.0k");
expect(html).not.toContain("Cost");
});
test("renders the model rows alone when the stage list carried no billing", () => {
const html = popoverMarkup(null);
expect(html).toContain("kimi-k3");
expect(html).not.toContain("Tokens");
});
});

View file

@ -67,6 +67,7 @@ import {
formatBytes,
formatDurationMs,
formatTokenCount,
formatUsdMicros,
} from "../lib/format";
import { plural } from "../lib/plural";
import {
@ -93,6 +94,7 @@ import {
type UnknownRecord,
} from "../lib/unknown";
import type {
BilledTokenCounts,
EventEnvelope,
ReasoningOutput,
StageHandler,
@ -866,10 +868,59 @@ export function formatStageModelUsageLabel(
return effort ? `${model}[${effort}]` : model;
}
function ModelUsagePopover({
const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
/**
* The disjoint token buckets behind a stage's usage, labelled and ordered to
* match the Billing tab's breakdown so the two views read the same. `Uncached`
* is input that missed the cache; `Output` folds in reasoning tokens.
*/
function stageTokenBuckets(billing: BilledTokenCounts) {
return [
{ label: "Cache read", value: billing.cache_read_tokens },
{ label: "Cache creation", value: billing.cache_write_tokens },
{ label: "Uncached", value: billing.input_tokens },
{
label: "Output",
value: billing.output_tokens + billing.reasoning_tokens,
},
];
}
/** Tokens and cost for this stage visit alone. */
function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
const buckets = stageTokenBuckets(billing);
if (buckets.every((bucket) => bucket.value === 0)) return null;
const cost = formatUsdMicros(billing.total_usd_micros);
return (
<div className="mt-3">
<PopoverHeader>Tokens</PopoverHeader>
<PopoverRows>
{buckets.map((bucket) => (
<PopoverRow key={bucket.label} label={bucket.label}>
<span className={POPOVER_NUMBER}>
{bucket.value === 0
? "0"
: formatTokenCount(bucket.value, { compactDecimal: true })}
</span>
</PopoverRow>
))}
{cost && (
<PopoverRow label="Cost">
<span className={POPOVER_NUMBER}>{cost}</span>
</PopoverRow>
)}
</PopoverRows>
</div>
);
}
export function ModelUsagePopover({
providerUsed,
billing,
}: {
providerUsed: StageModelUsage;
billing: BilledTokenCounts | null;
}) {
return (
<>
@ -892,6 +943,7 @@ function ModelUsagePopover({
<PopoverRow label="Speed">{providerUsed.speed}</PopoverRow>
)}
</PopoverRows>
{billing && <StageBillingRows billing={billing} />}
</>
);
}
@ -1905,6 +1957,7 @@ function EventsToolbar({
filteredCount,
totalCount,
providerUsed,
billing,
events,
runId,
stageId,
@ -1924,6 +1977,7 @@ function EventsToolbar({
filteredCount: number;
totalCount: number;
providerUsed: StageModelUsage | null;
billing: BilledTokenCounts | null;
events: EventEnvelope[];
runId: string;
stageId: string;
@ -2004,7 +2058,9 @@ function EventsToolbar({
className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${
showFilters ? "" : "ml-auto"
}`}
content={<ModelUsagePopover providerUsed={providerUsed} />}
content={
<ModelUsagePopover providerUsed={providerUsed} billing={billing} />
}
>
<CpuChipIcon className="size-3.5" aria-hidden="true" />
<span className="font-mono">{modelUsageLabel}</span>
@ -2359,6 +2415,7 @@ function RunStageActivityStage({
effectiveTab === "primary" ? turns.length : debugEvents.length
}
providerUsed={selectedStage.providerUsed}
billing={selectedStage.billing}
events={stageEventsQuery.data ?? []}
runId={runId}
stageId={selectedStageId}

View file

@ -12609,6 +12609,7 @@ components:
- status
- node_id
- visit
- billing
properties:
id:
$ref: "#/components/schemas/StageId"
@ -12668,6 +12669,15 @@ components:
format: date-time
description: Wall-clock time the latest attempt of this stage started, if known.
example: "2026-04-29T12:34:56Z"
billing:
$ref: "#/components/schemas/BilledTokenCounts"
description: >-
Token counts for this stage execution alone. `total_usd_micros` is
the provider-reported cost when there is one, otherwise the server
catalog's price for these tokens — the same pricing the
`/runs/{id}/billing` rows use. All-zero counts mean the stage made
no model calls. Unlike the billing rows, which sum every visit of a
node, this covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────

View file

@ -1134,6 +1134,7 @@ mod runs {
id: stage_id.clone(),
name: name.to_owned(),
handler,
billing: BilledTokenCounts::default(),
status,
wall_time_ms,
node_id: stage_id.node_id().to_owned(),

View file

@ -2,6 +2,7 @@ use std::collections::HashMap;
use std::sync::Arc;
use chrono::{DateTime, Utc};
use fabro_model::Catalog;
use fabro_types::{
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
};
@ -22,6 +23,7 @@ fn run_stage_from_projection(
stage_id: &StageId,
stage: &StageProjection,
graph: &Graph,
catalog: &Catalog,
now: DateTime<Utc>,
) -> RunStage {
let handler = stage.handler.unwrap_or_else(|| {
@ -36,6 +38,7 @@ fn run_stage_from_projection(
id: stage_id.clone(),
name: stage_id.node_id().to_owned(),
handler,
billing: stage.billed_usage(Some(catalog)).into_owned(),
status: stage.effective_state(),
wall_time_ms: stage.live_wall_time_ms(now),
node_id: stage_id.node_id().to_owned(),
@ -67,9 +70,10 @@ async fn list_run_stages(
let now = Utc::now();
let graph = projection.spec().graph();
let catalog = state.catalog();
let stages = projection
.iter_stages()
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, now))
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, &catalog, now))
.collect::<Vec<_>>();
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()

View file

@ -5791,6 +5791,147 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
assert_eq!(new_model["billing"]["input_tokens"], 200);
}
/// The stage popover reads `billing` off the stages list, so it must be scoped
/// to one visit — unlike the Billing tab's rows, which sum every visit of a
/// node. This exercises the same two-visit history as
/// `run_billing_sums_usage_across_retry_visits_and_uses_latest_model`.
#[tokio::test]
async fn list_run_stages_reports_billing_per_visit() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
1,
&workflow_event::Event::StageFailed {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(1200),
billing: Some(test_billed_usage("gpt-old", 100, 10)),
actor: None,
},
)
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
2,
&workflow_event::Event::StageCompleted {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
timing: fabro_types::StageTiming::wall_only(800),
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing: Some(test_billed_usage("gpt-new", 200, 20)),
failure: None,
notes: None,
files_touched: Vec::new(),
context_updates: None,
jump_to_node: None,
context_values: None,
node_visits: None,
loop_failure_signatures: None,
restart_failure_signatures: None,
response: None,
attempt: 2,
max_attempts: 2,
},
)
.await;
let response = app
.clone()
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let first = stage_entry(&body, "verify@1");
assert_eq!(first["billing"]["input_tokens"], 100);
assert_eq!(first["billing"]["output_tokens"], 10);
assert_eq!(first["billing"]["total_usd_micros"], 110);
let second = stage_entry(&body, "verify@2");
assert_eq!(second["billing"]["input_tokens"], 200);
assert_eq!(second["billing"]["output_tokens"], 20);
assert_eq!(second["billing"]["total_usd_micros"], 220);
}
#[tokio::test]
async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
&state,
run_id,
"script",
1,
&workflow_event::Event::StageStarted {
graph_visit: None,
resumed_from_stage_id: None,
node_id: "script".to_string(),
name: "Script".to_string(),
index: 0,
handler_type: "command".to_string(),
attempt: 1,
max_attempts: 1,
},
)
.await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let billing = &stage_entry(&body, "script@1")["billing"];
assert_eq!(billing["input_tokens"], 0);
assert_eq!(billing["output_tokens"], 0);
// No model ran, so there is nothing to price — not a $0.00 cost.
assert!(billing.get("total_usd_micros").is_none());
}
#[tokio::test]
async fn list_run_stages_shows_retrying_after_failed_event() {
let state = test_app_state_with_isolated_storage();

View file

@ -1,32 +1,7 @@
use std::borrow::Cow;
use std::collections::HashMap;
use fabro_model::Catalog;
use fabro_types::{
BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageTiming,
};
fn stage_usage_with_cost<'a>(
catalog: Option<&Catalog>,
stage: &'a StageProjection,
) -> Cow<'a, BilledTokenCounts> {
let Some(catalog) = catalog else {
return Cow::Borrowed(&stage.usage);
};
let Some(model) = stage.model.as_ref() else {
return Cow::Borrowed(&stage.usage);
};
if stage.usage.total_usd_micros.is_some() {
return Cow::Borrowed(&stage.usage);
}
let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
return Cow::Borrowed(&stage.usage);
};
let mut usage = stage.usage.clone();
usage.total_usd_micros = Some(total_usd_micros);
Cow::Owned(usage)
}
use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageTiming};
#[derive(Debug, Clone, PartialEq)]
pub struct ProjectionBillingStage {
@ -80,7 +55,7 @@ pub fn billing_rollup_from_projection(
if is_boundary_stage(projection, stage_id.node_id()) {
continue;
}
let usage = stage_usage_with_cost(catalog, stage);
let usage = stage.billed_usage(catalog);
let usage = usage.as_ref();
if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() {
continue;

View file

@ -3,7 +3,7 @@ use std::collections::{BTreeMap, HashMap};
use std::num::NonZeroU32;
use chrono::{DateTime, Utc};
use fabro_model::{ReasoningEffort, Speed};
use fabro_model::{Catalog, ReasoningEffort, Speed};
use strum::{Display, EnumString, IntoStaticStr};
use crate::run_event::{AgentSessionActivatedProps, StagePromptProps};
@ -540,6 +540,28 @@ impl StageProjection {
self.state
}
/// This stage's token counts with a cost attached.
///
/// A provider-reported cost always wins. Otherwise the catalog prices the
/// recorded tokens for the stage's model. The stored counts pass through
/// untouched when there is no catalog, no model, or no price for that
/// model, which leaves `total_usd_micros` as `None` rather than zero.
#[must_use]
pub fn billed_usage(&self, catalog: Option<&Catalog>) -> Cow<'_, BilledTokenCounts> {
if self.usage.total_usd_micros.is_some() {
return Cow::Borrowed(&self.usage);
}
let (Some(catalog), Some(model)) = (catalog, self.model.as_ref()) else {
return Cow::Borrowed(&self.usage);
};
let Some(total_usd_micros) = catalog.price_tokens(model, &self.usage.token_counts()) else {
return Cow::Borrowed(&self.usage);
};
let mut usage = self.usage.clone();
usage.total_usd_micros = Some(total_usd_micros);
Cow::Owned(usage)
}
/// Live wall-clock time in milliseconds.
///
/// While the stage is non-terminal (`Pending`, `Running`, or `Retrying`),
@ -844,11 +866,13 @@ mod iter_stages_tests {
use std::num::NonZeroU32;
use chrono::Utc;
use fabro_model::{Catalog, ModelRef, ProviderId};
use serde_json::json;
use super::RunProjection;
use crate::{
AgentControlState, Graph, RunId, RunSpec, StageProjection, WorkflowSettings, test_support,
AgentControlState, BilledTokenCounts, Graph, RunId, RunSpec, StageProjection,
WorkflowSettings, test_support,
};
fn seq(n: u32) -> NonZeroU32 {
@ -970,4 +994,62 @@ mod iter_stages_tests {
assert_eq!(order, vec!["build@1", "verify@1", "verify@2"]);
}
}
fn priced_stage(total_usd_micros: Option<i64>) -> StageProjection {
let mut stage = StageProjection::new(seq(1));
stage.usage = BilledTokenCounts {
input_tokens: 500_000,
output_tokens: 125_000,
total_tokens: 625_000,
total_usd_micros,
..BilledTokenCounts::default()
};
stage.model = Some(ModelRef {
provider: ProviderId::openai(),
model_id: "gpt-5.4".into(),
speed: None,
});
stage
}
#[test]
fn billed_usage_prices_uncosted_tokens_from_the_catalog() {
let stage = priced_stage(None);
assert_eq!(stage.billed_usage(None).total_usd_micros, None);
let priced = stage.billed_usage(Some(Catalog::builtin()));
assert!(
priced.total_usd_micros.is_some_and(|cost| cost > 0),
"expected a catalog price, got {:?}",
priced.total_usd_micros
);
// Pricing only fills in the cost; the token buckets pass through.
assert_eq!(priced.input_tokens, 500_000);
assert_eq!(priced.output_tokens, 125_000);
}
#[test]
fn billed_usage_keeps_a_provider_reported_cost_over_the_catalog_estimate() {
let stage = priced_stage(Some(42));
assert_eq!(
stage
.billed_usage(Some(Catalog::builtin()))
.total_usd_micros,
Some(42)
);
}
#[test]
fn billed_usage_leaves_a_modelless_stage_uncosted() {
let mut stage = priced_stage(None);
stage.model = None;
assert_eq!(
stage
.billed_usage(Some(Catalog::builtin()))
.total_usd_micros,
None
);
}
}

View file

@ -13,6 +13,9 @@
*/
// May contain unused imports in some cases
// @ts-ignore
import type { BilledTokenCounts } from './billed-token-counts';
// May contain unused imports in some cases
// @ts-ignore
import type { StageHandler } from './stage-handler';
@ -62,4 +65,8 @@ export interface RunStage {
* Wall-clock time the latest attempt of this stage started, if known.
*/
'started_at'?: string | null;
/**
* Token counts for this stage execution alone. `total_usd_micros` is the provider-reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/billing` rows use. All-zero counts mean the stage made no model calls. Unlike the billing rows, which sum every visit of a node, this covers only this visit.
*/
'billing': BilledTokenCounts;
}