diff --git a/apps/fabro-web/app/routes/run-billing.tsx b/apps/fabro-web/app/routes/run-billing.tsx
index 648b1796b..70fe0e5b2 100644
--- a/apps/fabro-web/app/routes/run-billing.tsx
+++ b/apps/fabro-web/app/routes/run-billing.tsx
@@ -97,6 +97,9 @@ function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
))}
+
+ Includes subagent tokens, priced at each subagent's model.
+
);
}
diff --git a/docs/internal/events.md b/docs/internal/events.md
index bcab1a858..add8b9411 100644
--- a/docs/internal/events.md
+++ b/docs/internal/events.md
@@ -422,6 +422,7 @@ Emitted when a workflow node finishes execution.
| `usage.reasoning_tokens` | number? | Reasoning/thinking tokens |
| `usage.speed` | string? | Speed tier |
| `usage.cost` | number? | Estimated cost in USD |
+| `billing_by_model` | array? | For an agent stage, the stage's billing split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know billed at the root's. Each row has `model`, `tokens`, and `total_usd_micros`, and the rows sum to the stage's billing. Empty for stages without a coding agent and on events written before it existed |
| `error` | string? | Error message (flattened from failure detail) |
| `failure_class` | string? | `"transient_infra"`, `"deterministic"`, `"budget_exhausted"`, `"compilation_loop"`, `"canceled"`, `"structural"` |
| `failure_signature` | string? | Dedup key for repeated failures |
@@ -433,6 +434,12 @@ Emitted when a workflow node finishes execution.
| `restart_failure_signatures` | object? | Restart failure signature counts |
| `response` | string? | Full LLM or agent response text when produced by the stage |
| `notes` | string? | Free-text notes |
+
+An agent stage's usage is its whole session tree's: the root session and
+every subagent, live in `StageProjection.usage` and here at completion, both
+read from the same fold of the stage's agent events. The root is priced at
+its route and each subagent at its own model; where the provider reported a
+cost, that cost stands.
| `files_touched` | string[] | File paths modified |
| `attempt` | number | Attempt number (1-based) |
| `max_attempts` | number | Maximum attempts allowed |
@@ -466,6 +473,8 @@ Emitted when a stage fails (before retry decision).
| `failure_class` | string | Failure category |
| `failure_signature` | string? | Dedup key for repeated failures |
| `will_retry` | boolean | Whether the stage will be retried |
+| `billing` | object? | What the stage spent before it failed, in the shape `stage.completed` uses. An agent stage that fails for good after answering model calls bills its whole session tree, as it would have on completion; a retried attempt and a cancelled stage carry none |
+| `billing_by_model` | array? | `billing` split by model, as on `stage.completed` |
### `stage.retrying`
diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml
index aa06923ed..3757ae924 100644
--- a/docs/public/api-reference/fabro-api.yaml
+++ b/docs/public/api-reference/fabro-api.yaml
@@ -11227,6 +11227,18 @@ components:
agent_control:
$ref: "#/components/schemas/AgentControlState"
description: Whether the agent is executing normally or waiting for steering after an interrupt.
+ billing_by_model:
+ type: array
+ items:
+ $ref: "#/components/schemas/BilledModelUsage"
+ default: []
+ description: >-
+ The completed stage's `usage` split by model, as `stage.completed`
+ reported it: the root session's route and each subagent's own
+ model, a subagent whose model the catalog does not know billed at
+ the root's. Sums to `usage`. Empty while the stage runs and for
+ stages without a coding agent; the billing rollup then bills
+ `usage` to `model`.
agent:
oneOf:
- $ref: "#/components/schemas/AgentSessionProjection"
@@ -13355,6 +13367,26 @@ components:
description: Billed USD amount in micros.
example: 720000
+ BilledModelUsage:
+ description: >-
+ Usage and cost billed to one model: one response, or one model's share
+ of a stage.
+ type: object
+ required:
+ - model
+ - tokens
+ properties:
+ model:
+ $ref: "#/components/schemas/BillingModelRef"
+ tokens:
+ $ref: "#/components/schemas/CompletionUsage"
+ total_usd_micros:
+ type: integer
+ format: int64
+ description: >-
+ Cost for `tokens`, when the provider reported one or the catalog
+ could price them. Absent means no cost data, not zero.
+
BillingModelRef:
description: Provider-qualified billing model identity used for cost estimates.
type: object
diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs
index 895494d78..7d9377dde 100644
--- a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs
+++ b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs
@@ -651,6 +651,7 @@ mod tests {
status: "succeeded".into(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs
index 94c6ce88c..ca3f5a020 100644
--- a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs
+++ b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs
@@ -650,6 +650,7 @@ mod tests {
status: "succeeded".into(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: Some(
billed_model_usage_from_llm(
&fabro_llm::test_support::test_catalog(),
diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs
index 1b86f0c90..2fd594239 100644
--- a/lib/apps/fabro-server/src/server/tests.rs
+++ b/lib/apps/fabro-server/src/server/tests.rs
@@ -6215,6 +6215,7 @@ fn stage_completed_event(node_id: &str) -> workflow_event::Event {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -7063,6 +7064,7 @@ async fn list_run_stages_projects_retrying_until_completion() {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -7102,14 +7104,15 @@ async fn list_run_stages_projects_retrying_until_completion() {
"work",
1,
&workflow_event::Event::StageFailed {
- node_id: "work".to_string(),
- name: "Work".to_string(),
- index: 1,
- failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(10),
- billing: None,
- actor: None,
+ node_id: "work".to_string(),
+ name: "Work".to_string(),
+ index: 1,
+ failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(10),
+ billing_by_model: Vec::new(),
+ billing: None,
+ actor: None,
},
)
.await;
@@ -7157,6 +7160,7 @@ async fn list_run_stages_projects_retrying_until_completion() {
status: "partially_succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -7370,14 +7374,15 @@ async fn create_billed_retry_run(state: &Arc, run_id: RunId) {
"verify",
1,
&workflow_event::Event::StageFailed {
- node_id: "verify".to_string(),
- name: "Verify".to_string(),
- index: 1,
- failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(1200),
- billing: Some(test_billed_usage("gpt-old", 100, 10)),
- actor: None,
+ node_id: "verify".to_string(),
+ name: "Verify".to_string(),
+ index: 1,
+ failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(1200),
+ billing_by_model: Vec::new(),
+ billing: Some(test_billed_usage("gpt-old", 100, 10)),
+ actor: None,
},
)
.await;
@@ -7394,6 +7399,7 @@ async fn create_billed_retry_run(state: &Arc, run_id: RunId) {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: Some(test_billed_usage("gpt-new", 200, 20)),
failure: None,
notes: None,
@@ -7482,6 +7488,7 @@ async fn list_run_stages_distinguishes_visits() {
status: "failed".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -7804,6 +7811,7 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() {
status: "failed".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -7835,6 +7843,7 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() {
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -8111,14 +8120,15 @@ async fn list_run_stages_shows_retrying_after_failed_event() {
"work",
1,
&workflow_event::Event::StageFailed {
- node_id: "work".to_string(),
- name: "Work".to_string(),
- index: 0,
- failure: FailureDetail::new("flake", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(5),
- billing: None,
- actor: None,
+ node_id: "work".to_string(),
+ name: "Work".to_string(),
+ index: 0,
+ failure: FailureDetail::new("flake", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(5),
+ billing_by_model: Vec::new(),
+ billing: None,
+ actor: None,
},
)
.await;
@@ -8193,14 +8203,15 @@ async fn list_run_stages_shows_retrying_when_failed_will_retry() {
"work",
1,
&workflow_event::Event::StageFailed {
- node_id: "work".to_string(),
- name: "Work".to_string(),
- index: 0,
- failure: FailureDetail::new("flake", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(5),
- billing: None,
- actor: None,
+ node_id: "work".to_string(),
+ name: "Work".to_string(),
+ index: 0,
+ failure: FailureDetail::new("flake", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(5),
+ billing_by_model: Vec::new(),
+ billing: None,
+ actor: None,
},
)
.await;
@@ -8243,14 +8254,15 @@ async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attemp
max_attempts: 3,
},
workflow_event::Event::StageFailed {
- node_id: "work".to_string(),
- name: "Work".to_string(),
- index: 0,
- failure: FailureDetail::new("transient", FailureCategory::TransientInfra),
- will_retry: true,
- timing: fabro_types::StageTiming::wall_only(10),
- billing: None,
- actor: None,
+ node_id: "work".to_string(),
+ name: "Work".to_string(),
+ index: 0,
+ failure: FailureDetail::new("transient", FailureCategory::TransientInfra),
+ will_retry: true,
+ timing: fabro_types::StageTiming::wall_only(10),
+ billing_by_model: Vec::new(),
+ billing: None,
+ actor: None,
},
workflow_event::Event::StageRetrying {
node_id: "work".to_string(),
@@ -8278,6 +8290,7 @@ async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attemp
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -8349,6 +8362,7 @@ fn revisit_test_completed_with_visit(
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -15402,6 +15416,7 @@ async fn active_acp_steerable_marker_clears_on_terminal_paths() {
status: "success".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
+ billing_by_model: Vec::new(),
billing: None,
failure: None,
notes: None,
@@ -15417,14 +15432,15 @@ async fn active_acp_steerable_marker_clears_on_terminal_paths() {
max_attempts: 1,
},
workflow_event::Event::StageFailed {
- node_id: "agent".to_string(),
- name: "agent".to_string(),
- index: 0,
- failure: FailureDetail::new("failed", FailureCategory::Deterministic),
- will_retry: false,
- timing: fabro_types::StageTiming::wall_only(1),
- billing: None,
- actor: None,
+ node_id: "agent".to_string(),
+ name: "agent".to_string(),
+ index: 0,
+ failure: FailureDetail::new("failed", FailureCategory::Deterministic),
+ will_retry: false,
+ timing: fabro_types::StageTiming::wall_only(1),
+ billing_by_model: Vec::new(),
+ billing: None,
+ actor: None,
},
];
diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs
index e0cedb00f..df6d0184b 100644
--- a/lib/components/fabro-store/src/run_state.rs
+++ b/lib/components/fabro-store/src/run_state.rs
@@ -25,7 +25,8 @@ use fabro_types::{
use fabro_util::error::render_compact_with_causes;
use lithos_llm::catalog::{ModelId, ProviderId};
use lithos_llm::types::TokenCounts;
-use pebble_coding_agent::events::{CodingEvent, TokenUsage};
+use pebble_coding_agent::events::CodingEvent;
+use pebble_coding_agent::projection::SessionProjection;
use crate::{Error, EventEnvelope, Result};
@@ -524,6 +525,7 @@ impl RunProjectionReducer for RunProjection {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
}
+ stage.billing_by_model.clone_from(&props.billing_by_model);
stage.state = StageState::from(outcome.status);
stage.agent_control = AgentControlState::Running;
}
@@ -547,6 +549,7 @@ impl RunProjectionReducer for RunProjection {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
}
+ stage.billing_by_model.clone_from(&props.billing_by_model);
stage.state =
stage_state_from_failure(props.will_retry, failure_category, stage.termination);
stage.agent_control = AgentControlState::Running;
@@ -760,9 +763,16 @@ fn apply_agent_event(
) {
let visit = props.visit;
// Pebble's own fold sees every agent event the stage stored, before the
- // fabro-only arms below read the same event.
+ // fabro-only arms below read the same event. While the stage runs, its
+ // usage is that fold's: the tree's tokens, the root's and every
+ // subagent's, with whatever cost the provider reported. The terminal
+ // billing then brings the catalog's price for the same tokens.
if let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) {
- stage.agent.get_or_insert_default().apply(&props.event);
+ let agent = stage.agent.get_or_insert_default();
+ agent.apply(&props.event);
+ if stage.completion.is_none() {
+ stage.usage = live_usage(agent);
+ }
}
#[expect(
clippy::wildcard_enum_match_arm,
@@ -771,17 +781,12 @@ fn apply_agent_event(
match props.coding_event() {
CodingEvent::AssistantMessage {
model,
- usage,
- cost_usd_micros,
context_window,
..
} => {
let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) else {
return;
};
- stage
- .usage
- .add_counts(&billed_counts(*usage, *cost_usd_micros));
if let Some(model) = stage_model_ref(stage, model) {
stage.model = Some(model);
}
@@ -984,11 +989,17 @@ fn apply_agent_event(
}
}
-/// Token accounting for one assistant message, in fabro's billing shape.
-fn billed_counts(usage: TokenUsage, cost_usd_micros: Option) -> BilledTokenCounts {
+/// A running stage's usage, from its agent's fold: the tree's tokens and the
+/// cost the provider reported for them, `None` when it reported none.
+fn live_usage(agent: &SessionProjection) -> BilledTokenCounts {
+ let (descendants, descendant_cost) = agent.descendant_usage();
+ let mut cost = agent.cost_usd_micros;
+ if let Some(descendant_cost) = descendant_cost {
+ cost = Some(cost.unwrap_or(0).saturating_add(descendant_cost));
+ }
BilledTokenCounts::from_token_counts(
- TokenCounts::from(usage),
- cost_usd_micros.map(|cost| i64::try_from(cost).unwrap_or(i64::MAX)),
+ TokenCounts::from(agent.usage.saturating_add(descendants)),
+ cost.map(|cost| i64::try_from(cost).unwrap_or(i64::MAX)),
)
}
@@ -1779,6 +1790,7 @@ fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome