Merge pull request #875 from fabro-sh/no-fabro-pricing
Some checks failed
Rust / Format (push) Waiting to run
Rust / Clippy (push) Waiting to run
Rust / Generated Docs (push) Waiting to run
Rust / Test (Linux) (push) Waiting to run
Rust / Sandbox plugins (stdio) (push) Waiting to run
Rust / Test (macOS) (push) Waiting to run
TypeScript / Typecheck (push) Has been cancelled
TypeScript / Test (push) Has been cancelled
TypeScript / Build (push) Has been cancelled

Take cost from the response: delete fabro's own catalog pricing
This commit is contained in:
Bryan Helmkamp 2026-09-14 17:06:42 -04:00 • committed by GitHub
commit cfd4843602
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
12 changed files with 400 additions and 407 deletions

View file

@ -33,8 +33,9 @@ export interface Stage {
startedAt: string | null;
providerUsed: StageModelUsage | null;
/**
* Tokens and cost for this visit alone, priced the same way the Usage tab
* prices its per-node rows. All-zero counts mean the stage called no model.
* Tokens and cost for this visit alone, the same figures the Usage tab
* sums into its per-node rows. All-zero counts mean the stage called no
* model.
*/
usage: Usage;
}

View file

@ -95,7 +95,7 @@ function TokenBreakdown({ usage }: { usage: Usage }) {
))}
</dl>
<p className="border-line text-fg-3 mt-1.5 border-t pt-1">
Includes subagent tokens, priced at each subagent&apos;s model.
Includes subagent tokens, each priced at the model it ran on.
</p>
</div>
);

View file

@ -412,7 +412,7 @@ Emitted when a workflow node finishes execution.
| `suggested_next_ids` | string[] | Suggested successor node ids |
| `usage` | object? | The stage's usage under the model it ran on (`ModelUsage`): for an agent stage, the whole session tree's tokens under the root's route. Absent for a stage that made no model calls |
| `usage.model` | object | `provider`, `model_id`, and optional `speed` tier |
| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost is the provider's reported figure when it gave one, else the catalog's price; absent when the catalog has no rates |
| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost sums what lithos-llm attached to each answer, the provider's reported figure when it gave one, else the catalog's price for the route; absent when an answer had neither |
| `usage_by_model` | array? | For an agent stage, `usage` split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know priced at the root's. Each row is a `ModelUsage`, and the rows sum to `usage`. Empty for stages without a coding agent and on events written before it existed |
| `error` | string? | Error message (flattened from failure detail) |
| `failure_class` | string? | `"transient_infra"`, `"deterministic"`, `"budget_exhausted"`, `"compilation_loop"`, `"canceled"`, `"structural"` |

View file

@ -11195,10 +11195,11 @@ components:
usage:
$ref: "#/components/schemas/Usage"
description: >-
The stage's usage: while the stage runs, its agent's own
accounting of the session tree with whatever cost the provider
reported; once it ends, the same tokens with the catalog's price
where the provider reported none.
The stage's usage: the session tree's tokens and cost, live and
once the stage ends. Every answer is priced once by lithos-llm
(the provider's reported cost, else the catalog's price for the
route) and summed; the cost is absent only when an answer had
neither.
model:
oneOf:
- $ref: "#/components/schemas/UsageModelRef"
@ -13839,12 +13840,12 @@ components:
usage:
$ref: "#/components/schemas/Usage"
description: >-
Usage for this stage execution alone. `cost` is the provider's
reported cost when there is one, otherwise the server catalog's
price for these tokens — the same pricing the `/runs/{id}/usage`
rows use. All-zero counts mean the stage made no model calls.
Unlike the usage rows, which sum every visit of a node, this
covers only this visit.
Usage for this stage execution alone. `cost` sums what lithos-llm
attached to each answer: the provider's reported cost when there
is one, otherwise the catalog's price for the route — the same
figures the `/runs/{id}/usage` rows sum. All-zero counts mean the
stage made no model calls. Unlike the usage rows, which sum every
visit of a node, this covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────

View file

@ -461,12 +461,12 @@ mod tests {
use fabro_workflow::event::{
Event, RunNoticeLevel, SandboxLifecycle, to_run_event, to_run_event_at,
};
use fabro_workflow::outcome::model_usage_from_llm;
use fabro_workflow::outcome::ModelUsage;
use lithos_llm::catalog::{ModelId, builtin};
use lithos_llm::types::TokenCounts;
use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage};
use pebble_coding_agent::events::{
CodingAgentEvent, CodingEvent, CompactionReason, ErrorData as AgentErrorData,
ErrorKind as AgentErrorKind, Usage,
ErrorKind as AgentErrorKind,
};
use super::*;
@ -649,18 +649,22 @@ mod tests {
preferred_label: None,
suggested_next_ids: Vec::new(),
usage_by_model: Vec::new(),
usage: Some(
model_usage_from_llm(
&fabro_llm::test_support::test_catalog(),
&ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")),
TokenCounts {
// Priced as lithos-llm prices gpt-5.4: 1200 input at $2.50/M and
// 300 output at $15/M.
usage: Some(ModelUsage::new(
ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")),
Usage {
tokens: TokenCounts {
input: 1200,
output: 300,
..TokenCounts::default()
},
)
.unwrap(),
),
cost: Some(Cost {
usd_micros: 7_500,
source: CostSource::Catalog,
}),
},
)),
failure: None,
notes: None,
files_touched: Vec::new(),

View file

@ -716,8 +716,8 @@ fn apply_agent_event(
// Pebble's own fold sees every agent event the stage stored, before the
// fabro-only arms below read the same event. While the stage runs, its
// usage is that fold's: the tree's tokens, the root's and every
// subagent's, with whatever cost the provider reported. The terminal
// usage then brings the catalog's price for the same tokens.
// subagent's, with the cost lithos-llm attached to each answer. The
// terminal usage is the same sum, split by model.
if let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) {
let agent = stage.agent.get_or_insert_default();
agent.apply(&props.event);
@ -5342,11 +5342,105 @@ mod tests {
}
/// One usage rule: a stage's usage is its session tree's, live and at
/// completion. The terminal usage carries the tokens the fold already
/// showed plus the catalog's price, so completion changes the cost, not
/// the tokens, and keeps the split by model.
/// completion, cost included. lithos-llm prices each answer once, pebble
/// sums them, and the terminal usage is the same sum, so completion
/// changes neither the tokens nor the cost; it adds the split by model.
#[test]
fn stage_completed_keeps_the_trees_live_usage_and_prices_it() {
fn stage_completed_keeps_the_trees_live_usage_and_its_cost() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let model = priced_usage().model().clone();
let catalog_priced = |input: u64, output: u64, usd_micros: u64| Usage {
tokens: TokenCounts {
input,
output,
..TokenCounts::default()
},
cost: Some(Cost {
usd_micros,
source: CostSource::Catalog,
}),
};
let message = |session: &str, usage: Usage| {
let mut event = CodingAgentEvent::new(
session,
CodingEvent::AssistantMessage {
text: "assistant text".to_string(),
model: model.model_id.to_string(),
usage,
tool_call_count: 0,
context_window: None,
reasoning: None,
},
SystemTime::UNIX_EPOCH,
);
if session != "ses_test" {
event = event.with_parent_session_id("ses_test");
}
EventBody::Agent(AgentEventProps::new("code", 1, event))
};
state
.apply_event(&test_stage_event(
1,
EventBody::StageStarted(started_props()),
stage_id.clone(),
))
.unwrap();
state
.apply_event(&test_stage_event(
2,
activated(model.provider.as_str(), model.model_id.as_str()),
stage_id.clone(),
))
.unwrap();
state
.apply_event(&test_stage_event(
3,
message("ses_test", catalog_priced(100, 50, 300)),
stage_id.clone(),
))
.unwrap();
state
.apply_event(&test_stage_event(
4,
message("ses_child", catalog_priced(7, 1, 21)),
stage_id.clone(),
))
.unwrap();
let live = state.stage(&stage_id).unwrap().usage;
assert_eq!(
live,
catalog_priced(107, 51, 321),
"the subagent's tokens and cost are the stage's too"
);
// The terminal usage is the same sum, under the root's route.
let tree = ModelUsage::new(model.clone(), live);
let mut props = completed_props(42, StageOutcome::Succeeded);
props.usage = Some(tree.clone());
props.usage_by_model = vec![tree.clone()];
state
.apply_event(&test_stage_event(
5,
EventBody::StageCompleted(props),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.usage, live,
"completion keeps the usage the fold showed, cost included"
);
assert_eq!(stage.model.as_ref(), Some(&model));
assert_eq!(stage.usage_by_model, vec![tree]);
}
/// An answer nobody priced leaves the tree's cost unknown, live and at
/// completion alike; the tokens are still counted.
#[test]
fn an_unpriced_answer_leaves_the_stage_cost_unknown_live_and_at_completion() {
let mut state = initialized_projection();
let stage_id = StageId::new("build", 1);
let model = priced_usage().model().clone();
@ -5372,57 +5466,28 @@ mod tests {
stage_id.clone(),
))
.unwrap();
let live = state.stage(&stage_id).unwrap().usage;
assert_eq!(live, live_counts(100, 50));
assert_eq!(live.cost, None);
let tree = ModelUsage::new(model.clone(), live);
let mut props = completed_props(42, StageOutcome::Succeeded);
props.usage = Some(tree.clone());
props.usage_by_model = vec![tree];
state
.apply_event(&test_stage_event(
4,
child_message_body(7, 1),
stage_id.clone(),
))
.unwrap();
let live = state.stage(&stage_id).unwrap().usage;
assert_eq!(
live,
live_counts(107, 51),
"the subagent's tokens are the stage's too"
);
let tree = ModelUsage::new(model.clone(), Usage {
tokens: TokenCounts {
input: 107,
output: 51,
..TokenCounts::default()
},
cost: Some(Cost {
usd_micros: 321,
source: CostSource::Catalog,
}),
});
let mut props = completed_props(42, StageOutcome::Succeeded);
props.usage = Some(tree.clone());
props.usage_by_model = vec![tree.clone()];
state
.apply_event(&test_stage_event(
5,
EventBody::StageCompleted(props),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(stage.usage, live);
assert_eq!(
stage.usage.tokens, live.tokens,
"completion keeps the tokens the fold showed"
stage.usage.cost, None,
"nothing priced it, so nothing invents a cost"
);
assert_eq!(
stage.usage.cost,
Some(Cost {
usd_micros: 321,
source: CostSource::Catalog,
}),
"and brings the catalog's price"
);
assert_eq!(stage.model.as_ref(), Some(&model));
assert_eq!(stage.usage_by_model, vec![tree]);
}
#[test]

View file

@ -63,7 +63,7 @@ use crate::context::keys::Fidelity;
use crate::error::Error;
use crate::event::{Emitter, Event, StageScope};
use crate::model_fallback::{ModelFallbackNotice, ModelFallbackPolicy};
use crate::outcome::{Outcome, model_usage_from_llm, with_reported_cost};
use crate::outcome::Outcome;
use crate::services::FabroRunToolServices;
use crate::steering_hub::SteeringHub;
use crate::web_search::{self, SearchSecrets};
@ -336,20 +336,18 @@ struct StageUsage {
by_model: Vec<ModelUsage>,
}
/// Prices the stage's account from the catalog: the root session at
/// `root_model`, its route, and each descendant at its own route where the
/// catalog knows it and at the root's otherwise, so a subagent on a cheaper
/// or dearer model is priced as what it ran. A descendant on the root's
/// route joins the root's row. Where pebble carried a provider-reported
/// cost, that cost stands in for the catalog's estimate. The total's cost is
/// the rows' sum, which is `None` once a row that used tokens has no cost.
/// The stage's account grouped by model: the root session at `root_model`,
/// its route, and each descendant at its own route where the catalog knows
/// it and at the root's otherwise. A descendant on the root's route joins
/// the root's row. Every cost is the one pebble carried: lithos-llm attaches
/// the provider's reported cost or the catalog's price to each answer, and
/// pebble sums them per session, so fabro prices nothing of its own. A row,
/// and the total, has a cost only when every answer in it was priced.
fn stage_usage(
catalog: &Catalog,
root_model: &ModelRef,
account: &SessionProjection,
) -> Result<StageUsage, Error> {
// Each group's usage is the sum of pebble's accounts, so its cost is what
// the provider reported, or `None` once an unpriced account is in it.
) -> StageUsage {
let mut groups: Vec<(ModelRef, Usage)> = vec![(root_model.clone(), account.usage)];
for descendant in account.descendants.values() {
let model = descendant_model(catalog, root_model, descendant);
@ -361,25 +359,20 @@ fn stage_usage(
// The root's row first, then the others by model.
groups[1..].sort_by(|left, right| left.0.sort_key().cmp(&right.0.sort_key()));
let mut by_model = Vec::with_capacity(groups.len());
let mut total = Usage::default();
for (model, usage) in groups {
let row = with_reported_cost(
model_usage_from_llm(catalog, &model, usage.tokens)?,
usage.cost,
);
total = total.saturating_add(row.usage);
by_model.push(row);
let total = fabro_types::sum_usage(groups.iter().map(|(_, usage)| *usage));
StageUsage {
total: ModelUsage::new(root_model.clone(), total),
by_model: groups
.into_iter()
.map(|(model, usage)| ModelUsage::new(model, usage))
.collect(),
}
Ok(StageUsage {
total: ModelUsage::new(root_model.clone(), total),
by_model,
})
}
/// The route a descendant is billed at: its own where its start named one
/// the catalog knows, else the root's. A descendant whose start was not seen
/// names only its answers' model, taken to be on the root's provider.
/// The route a descendant's usage is grouped under: its own where its start
/// named one the catalog knows, else the root's. A descendant whose start
/// was not seen names only its answers' model, taken to be on the root's
/// provider.
fn descendant_model(
catalog: &Catalog,
root_model: &ModelRef,
@ -756,27 +749,17 @@ impl PebbleBackend {
/// The failed outcome of an agent stage that spent before it failed: the
/// failure itself, with the session tree's usage, the files it wrote, and
/// its active time, so the run records what the stage spent. A usage the
/// catalog cannot price is logged and left off.
/// its active time, so the run records what the stage spent.
fn failed_outcome(&self, error: &Error, live: &LiveAgent, plan: &FallbackPlan) -> Outcome {
let mut outcome = error.to_fail_outcome();
let account = live.account();
match stage_usage(
let usage = stage_usage(
self.catalog.as_ref(),
&route_model(plan.current()),
&account,
) {
Ok(usage) => {
outcome.usage = Some(usage.total);
outcome.usage_by_model = usage.by_model;
}
Err(usage_error) => {
tracing::debug!(
error = %usage_error,
"failed agent stage could not be priced"
);
}
}
);
outcome.usage = Some(usage.total);
outcome.usage_by_model = usage.by_model;
outcome.files_touched = account.files_touched;
outcome.timing = Some(StageTiming::active_only(
crate::millis_u64(live.inference_duration),
@ -1028,12 +1011,10 @@ impl CodergenBackend for PebbleBackend {
continue;
}
// The provider's own cost, when every answer carried one, stands in
// for the catalog's estimate.
let stage_usage = with_reported_cost(
model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage.tokens)?,
total_usage.cost,
);
// Each response came priced by lithos-llm: the provider's reported
// cost, or the catalog's price for the route. The stage's cost is
// their sum, known only when every answer was priced.
let stage_usage = ModelUsage::new(completion.model.clone(), total_usage);
return Ok(CodergenResult::Text {
text: response_text,
@ -1238,7 +1219,7 @@ impl CodergenBackend for PebbleBackend {
self.catalog.as_ref(),
&route_model(fallback_plan.current()),
&account,
)?;
);
live.release_lease();
match reuse_key {
@ -1303,7 +1284,7 @@ mod tests {
}
}
fn message(model: &str, input: u64, output: u64, cost: Option<u64>) -> CodingEvent {
fn message(model: &str, input: u64, output: u64, cost: Option<Cost>) -> CodingEvent {
CodingEvent::AssistantMessage {
text: "ok".to_string(),
model: model.to_string(),
@ -1313,10 +1294,7 @@ mod tests {
output,
..TokenCounts::default()
},
cost: cost.map(|usd_micros| Cost {
usd_micros,
source: CostSource::Provider,
}),
cost,
},
tool_call_count: 0,
context_window: None,
@ -1324,6 +1302,13 @@ mod tests {
}
}
fn catalog_cost(usd_micros: u64) -> Cost {
Cost {
usd_micros,
source: CostSource::Catalog,
}
}
fn root_model() -> ModelRef {
ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4"))
}
@ -1334,8 +1319,10 @@ mod tests {
account
}
/// Every cost comes from pebble's stream, where lithos-llm attached it
/// to each answer; fabro groups and sums, and prices nothing itself.
#[test]
fn stage_usage_prices_the_root_at_its_route_and_each_descendant_at_its_own() {
fn stage_usage_groups_pebbles_priced_accounts_by_route_and_sums_them() {
let catalog = test_catalog();
let account = account(&[
root(started("openai", "gpt-5.4")),
@ -1344,20 +1331,43 @@ mod tests {
content: None,
source: InputSource::Prompt,
}),
root(message("gpt-5.4", 100_000, 25_000, None)),
root(message(
"gpt-5.4",
100_000,
25_000,
Some(catalog_cost(300_000)),
)),
// A child on the parent's route joins the parent's row.
child("ses_same", started("openai", "gpt-5.4")),
child("ses_same", message("gpt-5.4", 10_000, 1_000, None)),
// A child on another route is its own row, at that route's rate.
child(
"ses_same",
message("gpt-5.4", 10_000, 1_000, Some(catalog_cost(30_000))),
),
// A child on another route is its own row, at the cost its
// provider reported.
child("ses_other", started("anthropic", "claude-sonnet-5")),
child("ses_other", message("claude-sonnet-5", 20_000, 2_000, None)),
// A child on a route the catalog does not know bills at the root's.
child(
"ses_other",
message(
"claude-sonnet-5",
20_000,
2_000,
Some(Cost {
usd_micros: 70_000,
source: CostSource::Provider,
}),
),
),
// A child on a route the catalog does not know joins the root's row.
child("ses_unknown", started("nowhere", "mystery")),
child("ses_unknown", message("mystery", 1_000, 100, None)),
child(
"ses_unknown",
message("mystery", 1_000, 100, Some(catalog_cost(5_000))),
),
root(CodingEvent::ProcessingEnd),
]);
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
let usage = stage_usage(&catalog, &root_model(), &account);
assert_eq!(usage.by_model.len(), 2, "{:?}", usage.by_model);
let root_row = &usage.by_model[0];
@ -1367,102 +1377,103 @@ mod tests {
"the root, the same-route child, and the unknown-route child"
);
assert_eq!(root_row.usage.tokens.output, 26_100);
let root_priced =
model_usage_from_llm(&catalog, &root_model(), root_row.usage.tokens).unwrap();
assert_eq!(root_row.usage.cost, root_priced.usage.cost);
assert_eq!(
root_row.usage.cost.map(|cost| cost.source),
Some(CostSource::Catalog)
root_row.usage.cost,
Some(catalog_cost(335_000)),
"the row's cost is the sum of what pebble carried, still the catalog's"
);
let other_model = ModelRef::new(
ProviderId::new("anthropic"),
ModelId::new("claude-sonnet-5"),
);
let other_row = &usage.by_model[1];
assert_eq!(other_row.model, other_model);
assert_eq!(
other_row.model,
ModelRef::new(
ProviderId::new("anthropic"),
ModelId::new("claude-sonnet-5"),
)
);
assert_eq!(other_row.usage.tokens.input, 20_000);
assert_eq!(other_row.usage.tokens.output, 2_000);
let other_priced =
model_usage_from_llm(&catalog, &other_model, other_row.usage.tokens).unwrap();
assert_eq!(other_row.usage.cost, other_priced.usage.cost);
assert_ne!(
assert_eq!(
other_row.usage.cost,
model_usage_from_llm(&catalog, &root_model(), other_row.usage.tokens)
.unwrap()
.usage
.cost,
"priced at its own rate, not the root's"
Some(Cost {
usd_micros: 70_000,
source: CostSource::Provider,
}),
"a provider-reported cost is kept as reported"
);
// The total is the tree's tokens under the root's route, at the rows' summed
// cost, from the catalog like every row.
// The total is the tree's tokens under the root's route; its cost is
// the rows' sum, assembled from two sources.
assert_eq!(usage.total.model, root_model());
assert_eq!(usage.total.usage.tokens.input, 131_000);
assert_eq!(usage.total.usage.tokens.output, 28_100);
assert_eq!(
usage.total.usage.cost,
Some(Cost {
usd_micros: root_priced.usage.cost.unwrap().usd_micros
+ other_priced.usage.cost.unwrap().usd_micros,
source: CostSource::Catalog,
})
);
}
#[test]
fn a_provider_reported_cost_stands_in_for_the_catalogs_estimate() {
let catalog = test_catalog();
let account = account(&[
root(started("openai", "gpt-5.4")),
root(message("gpt-5.4", 1_000, 100, Some(4_321))),
child("ses_child", started("anthropic", "claude-sonnet-5")),
child("ses_child", message("claude-sonnet-5", 500, 50, None)),
]);
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
assert_eq!(
usage.by_model[0].usage.cost,
Some(Cost {
usd_micros: 4_321,
source: CostSource::Provider,
})
);
let child_priced = model_usage_from_llm(
&catalog,
&usage.by_model[1].model,
usage.by_model[1].usage.tokens,
)
.unwrap();
assert_eq!(usage.by_model[1].usage.cost, child_priced.usage.cost);
// One row reported, one priced: the sum is the application's.
assert_eq!(
usage.total.usage.cost,
Some(Cost {
usd_micros: 4_321 + child_priced.usage.cost.unwrap().usd_micros,
usd_micros: 405_000,
source: CostSource::Application,
})
);
}
/// An answer pebble could not price (a model with no catalog price and
/// no provider cost) leaves its row's cost, and the total's, unknown; the
/// tokens are still counted. Live and completed usage agree because both
/// are the same sum of pebble's accounts.
#[test]
fn a_descendant_seen_only_through_its_answers_bills_on_the_roots_provider() {
fn stage_usage_leaves_the_cost_unknown_once_an_answer_was_unpriced() {
let catalog = test_catalog();
let priced_only = account(&[
root(started("openai", "gpt-5.4")),
root(message("gpt-5.4", 1_000, 100, Some(catalog_cost(4_321)))),
]);
let priced = stage_usage(&catalog, &root_model(), &priced_only);
assert_eq!(priced.total.usage.cost, Some(catalog_cost(4_321)));
assert_eq!(
priced.total.usage,
priced_only
.usage
.saturating_add(priced_only.descendant_usage()),
"the completed usage is the live fold's, cost included"
);
let tree = account(&[
root(started("openai", "gpt-5.4")),
root(message("gpt-5.4", 1_000, 100, Some(catalog_cost(4_321)))),
child("ses_child", started("anthropic", "claude-sonnet-5")),
child("ses_child", message("claude-sonnet-5", 500, 50, None)),
]);
let usage = stage_usage(&catalog, &root_model(), &tree);
assert_eq!(usage.by_model[0].usage.cost, Some(catalog_cost(4_321)));
assert_eq!(usage.by_model[1].usage.tokens.input, 500);
assert_eq!(usage.by_model[1].usage.cost, None);
assert_eq!(usage.total.usage.tokens.input, 1_500);
assert_eq!(usage.total.usage.cost, None);
assert_eq!(
usage.total.usage,
tree.usage.saturating_add(tree.descendant_usage()),
"the completed usage is the live fold's, cost unknown at both"
);
}
#[test]
fn a_descendant_seen_only_through_its_answers_groups_under_the_roots_provider() {
let catalog = test_catalog();
let mut account = account(&[root(started("openai", "gpt-5.4"))]);
// No `SessionStarted` for the child: only its answer names a model.
account.apply(&child(
"ses_quiet",
message("gpt-5.4-mini", 1_000, 100, None),
message("gpt-5.4-mini", 1_000, 100, Some(catalog_cost(1))),
));
let usage = stage_usage(&catalog, &root_model(), &account).unwrap();
let usage = stage_usage(&catalog, &root_model(), &account);
let child_row = usage
.by_model
.iter()
.find(|row| row.model.model_id.as_str() == "gpt-5.4-mini")
.expect("the child is billed as its answers' model on the root's provider");
.expect("the child is grouped as its answers' model on the root's provider");
assert_eq!(child_row.model.provider, root_model().provider);
assert_eq!(child_row.usage.tokens.input, 1_000);
}

View file

@ -591,22 +591,20 @@ mod tests {
use fabro_graphviz::graph::AttrValue;
use fabro_types::ModelRef;
use lithos_llm::catalog::{ModelId, builtin};
use lithos_llm::types::TokenCounts;
use lithos_llm::types::{TokenCounts, Usage};
use super::*;
use crate::outcome::{ModelUsage, model_usage_from_llm};
use crate::outcome::ModelUsage;
fn stage_usage(model: &str, input: u64, output: u64) -> ModelUsage {
model_usage_from_llm(
&fabro_llm::test_support::test_catalog(),
&ModelRef::new(builtin::anthropic(), ModelId::new(model)),
TokenCounts {
ModelUsage::new(
ModelRef::new(builtin::anthropic(), ModelId::new(model)),
Usage::from(TokenCounts {
input,
output,
..TokenCounts::default()
},
}),
)
.unwrap()
}
fn large_prompt_value(bytes: u64, path: &str, preview: &str) -> serde_json::Value {

View file

@ -1,46 +1,12 @@
pub use fabro_core::outcome::{
FailureCategory, FailureDetail, OutcomeMeta, StageOutcome, StageState,
};
use fabro_llm::lithos_catalog::Catalog;
use fabro_types::ModelRef;
pub use fabro_types::ModelUsage;
use lithos_llm::types::{Cost, TokenCounts, Usage};
use crate::error::{Error, FailureSignature, classify_failure_reason};
use crate::error::{FailureSignature, classify_failure_reason};
pub type Outcome = fabro_core::Outcome<Option<ModelUsage>>;
/// Prices `tokens` on `model` from the catalog: the usage carries a
/// [`CostSource::Catalog`](lithos_llm::types::CostSource::Catalog) cost when
/// the catalog has rates for the model, and no cost otherwise.
///
/// The provider must be one the catalog knows; a passthrough model on a known
/// provider is priced with no cost, since the catalog has no rates for it.
pub fn model_usage_from_llm(
catalog: &Catalog,
model: &ModelRef,
tokens: TokenCounts,
) -> Result<ModelUsage, Error> {
if catalog.enabled_provider(model.provider.as_str()).is_none() {
return Err(Error::Precondition(format!(
"Provider \"{}\" is not configured",
model.provider
)));
}
let cost = catalog.estimate_cost(&model.handle(), tokens, model.speed);
Ok(ModelUsage::new(model.clone(), Usage { tokens, cost }))
}
/// `usage` with `cost` in place of whatever it carried, when a provider
/// reported one; `None` keeps the usage as it is.
#[must_use]
pub fn with_reported_cost(mut usage: ModelUsage, cost: Option<Cost>) -> ModelUsage {
if let Some(cost) = cost {
usage.usage.cost = Some(cost);
}
usage
}
pub trait OutcomeExt: Sized {
fn fail_deterministic(reason: impl Into<String>) -> Self;
fn fail_classify(reason: impl Into<String>) -> Self;
@ -131,75 +97,7 @@ pub fn format_cost(cost: f64) -> String {
#[cfg(test)]
mod tests {
use fabro_llm::lithos_catalog::Catalog;
use fabro_llm::test_support::{test_catalog, test_catalog_with_overlay};
use fabro_types::ModelRef;
use lithos_llm::catalog::{ModelId, ProviderId, builtin};
use lithos_llm::types::{Cost, CostSource, Speed, TokenCounts};
use super::{OutcomeExt, model_usage_from_llm, with_reported_cost};
fn model_ref(provider: ProviderId, model_id: &str, speed: Option<Speed>) -> ModelRef {
ModelRef::new(provider, ModelId::new(model_id)).with_speed(speed)
}
fn catalog() -> Catalog {
test_catalog()
}
#[test]
fn model_usage_from_llm_prices_openai_cached_input_and_reasoning_output() {
// Stay under the 272k long-context tier so the standard rates apply.
let usage = TokenCounts {
input: 100_000,
output: 25_000,
reasoning: 5_000,
cache_read: 50_000,
..TokenCounts::default()
};
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "gpt-5.4", None),
usage,
)
.unwrap();
// 100k input at $2.50/M + 50k cached at $0.25/M + 30k output at $15/M.
assert_eq!(
billed.usage.cost,
Some(Cost {
usd_micros: 712_500,
source: CostSource::Catalog,
})
);
assert_eq!(billed.usage.tokens.output, 25_000);
assert_eq!(billed.usage.tokens.reasoning, 5_000);
}
#[test]
fn response_cost_overrides_catalog_estimate() {
let usage = TokenCounts {
input: 11,
output: 7,
..TokenCounts::default()
};
let reported = Cost {
usd_micros: 125_000,
source: CostSource::Provider,
};
let billed = with_reported_cost(
model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "gpt-5.4", None),
usage,
)
.unwrap(),
Some(reported),
);
assert_eq!(billed.usage.cost, Some(reported));
assert_eq!(billed.usage.tokens, usage);
}
use super::OutcomeExt;
#[test]
fn retry_classify_marks_failed_outcome_with_retry_request() {
@ -210,94 +108,4 @@ mod tests {
});
assert!(outcome.status.retry_requested());
}
#[test]
fn model_usage_from_llm_prices_anthropic_fast_mode_cache_write_rates() {
let usage = TokenCounts {
input: 100_000,
output: 10_000,
reasoning: 5_000,
cache_read: 20_000,
cache_write: 30_000,
};
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::anthropic(), "claude-opus-5", Some(Speed::Fast)),
usage,
)
.unwrap();
// Fast rates: $10/M input, $50/M output (incl. reasoning), $1/M cache
// read, $12.50/M cache write.
assert_eq!(
billed.usage.cost.map(|cost| cost.usd_micros),
Some(2_145_000)
);
}
#[test]
fn model_usage_from_llm_uses_injected_custom_catalog() {
let catalog = test_catalog_with_overlay(
r#"
[providers.proxy]
display_name = "Proxy"
adapter = "openai-compatible"
codec = "openai-chat"
base_url = "https://proxy.example/v1"
auth = { type = "bearer" }
default_model = "canonical-model"
[providers.proxy.models.canonical-model]
display_name = "Canonical Model"
api_model = "wire-model"
limits = { context_tokens = 1000, max_output_tokens = 500 }
capabilities = { text = true, tools = true }
pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_million = 2000000 }
"#,
);
let usage = TokenCounts {
input: 1_000_000,
output: 500_000,
..TokenCounts::default()
};
let billed = model_usage_from_llm(
&catalog,
&model_ref(ProviderId::new("proxy"), "canonical-model", None),
usage,
)
.unwrap();
assert_eq!(
billed.usage.cost.map(|cost| cost.usd_micros),
Some(2_000_000)
);
assert_eq!(billed.model_id(), "canonical-model");
}
#[test]
fn passthrough_model_on_known_provider_has_no_cost() {
let billed = model_usage_from_llm(
&catalog(),
&model_ref(builtin::openai(), "brand-new-model", None),
TokenCounts {
input: 10,
output: 5,
..TokenCounts::default()
},
)
.unwrap();
assert_eq!(billed.usage.cost, None);
assert_eq!(billed.usage.tokens.input, 10);
}
#[test]
fn unknown_provider_is_a_precondition_failure() {
let error = model_usage_from_llm(
&catalog(),
&model_ref(ProviderId::new("nowhere"), "model", None),
TokenCounts::default(),
)
.unwrap_err();
assert!(error.to_string().contains("not configured"), "{error}");
}
}

View file

@ -44,6 +44,7 @@ use fabro_workflow::test_support::WorkflowRunner;
use httpmock::Method::POST;
use httpmock::MockServer;
use lithos_llm::catalog::ProviderId;
use lithos_llm::types::{Cost, CostSource};
use pebble_coding_agent::events::{CodingEvent, FailoverContinuation, FailoverStop};
use tokio_util::sync::CancellationToken;
@ -235,6 +236,19 @@ fn agent_registry(backend: PebbleBackend) -> HandlerRegistry {
registry
}
fn prompt_registry(backend: PebbleBackend) -> HandlerRegistry {
let mut registry = HandlerRegistry::new(Box::new(StartHandler));
registry.register("start", Box::new(StartHandler));
registry.register("exit", Box::new(ExitHandler));
registry.register(
"prompt",
Box::new(fabro_workflow::handler::prompt::PromptHandler::new(Some(
Box::new(backend),
))),
);
registry
}
/// Every run event the run emitted, in order.
type Events = Arc<Mutex<Vec<RunEvent>>>;
@ -339,6 +353,25 @@ impl Stage {
state
}
/// Runs `graph` with `work` as a one-shot prompt stage and returns the
/// projection.
async fn run_prompt_ok(
&self,
backend: PebbleBackend,
graph: &Graph,
) -> fabro_types::RunProjection {
let sandbox = local_sandbox(self.dir.path()).await;
let runner =
WorkflowRunner::new(prompt_registry(backend), Arc::clone(&self.emitter), sandbox);
let options = run_options(self.dir.path(), CancellationToken::new());
let (outcome, state) = runner
.run_with_state(graph, &options)
.await
.expect("workflow execution should complete");
assert_eq!(outcome.status, StageOutcome::Succeeded, "{outcome:?}");
state
}
/// Fires `action` once, when the stage's first model call starts.
fn on_first_llm_call(&self, action: impl Fn() + Send + Sync + 'static) {
let fired = AtomicBool::new(false);
@ -408,7 +441,7 @@ async fn write_file_under_profile(profile: &str, tool: &str, path_key: &str) {
assert_eq!(
work.usage.cost.map(|cost| cost.usd_micros),
Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)),
"{profile}: cost from the catalog's pricing"
"{profile}: every answer came priced from the catalog"
);
let checkpoint = state.current_checkpoint().expect("a checkpoint");
let outcome = checkpoint
@ -831,7 +864,7 @@ async fn a_stage_that_fails_after_spending_bills_what_it_spent() {
assert_eq!(
work.usage.cost.map(|cost| cost.usd_micros),
Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)),
"priced from the catalog like a completed stage"
"every answer came priced from the catalog"
);
assert_eq!(work.usage_by_model.len(), 1, "{:?}", work.usage_by_model);
assert_eq!(
@ -1006,7 +1039,7 @@ async fn a_subagent_runs_under_its_parent_session() {
assert_eq!(
work.usage.cost.map(|cost| cost.usd_micros),
Some(4 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)),
"priced from the catalog for every call"
"every answer came priced from the catalog"
);
let agent = work
.agent
@ -1014,9 +1047,14 @@ async fn a_subagent_runs_under_its_parent_session() {
.expect("the stage carries pebble's fold");
let descendants = agent.descendant_usage();
assert_eq!(
work.usage.tokens.input,
agent.usage.tokens.input + descendants.tokens.input,
"the completed usage is what the live fold showed"
work.usage,
agent.usage.saturating_add(descendants),
"the completed usage is what the live fold showed, cost included"
);
assert_eq!(
work.usage.cost.map(|cost| cost.source),
Some(CostSource::Catalog),
"lithos-llm priced every answer from the catalog; fabro priced nothing"
);
assert_eq!(descendants.tokens.input, INPUT_TOKENS_PER_CALL);
// The child ran on its parent's model, so the split is one row carrying
@ -1613,3 +1651,70 @@ async fn daytona_sandbox_runs_an_agent_stage() {
sandbox.delete().await.expect("Daytona cleanup failed");
}
// --- One-shot prompt stages --------------------------------------------------
/// A one-shot prompt stage calls lithos-llm's client directly, and the
/// response comes back priced: the resolver fills the catalog's price for
/// the route when the provider reported none. Fabro records that cost as
/// is; it estimates nothing itself.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn a_prompt_stage_records_the_catalog_cost_lithos_attached_to_the_response() {
let stage = Stage::new().await;
let completion = serde_json::json!({
"id": "chatcmpl-prompt",
"object": "chat.completion",
"model": MODEL,
"choices": [{
"index": 0,
"message": { "role": "assistant", "content": "Summarized." },
"finish_reason": "stop"
}],
"usage": {
"prompt_tokens": INPUT_TOKENS_PER_CALL,
"completion_tokens": OUTPUT_TOKENS_PER_CALL,
"total_tokens": INPUT_TOKENS_PER_CALL + OUTPUT_TOKENS_PER_CALL,
}
});
let mock = stage
.server
.mock_async(|when, then| {
when.method(POST)
.path(CHAT_PATH)
.body_includes("Summarize the change");
then.status(200)
.header("content-type", "application/json")
.body(completion.to_string());
})
.await;
let mut graph = agent_graph("Prompt", "Summarize the change");
graph
.nodes
.get_mut("work")
.expect("the work node")
.attrs
.insert("type".to_string(), AttrValue::String("prompt".to_string()));
let state = stage.run_prompt_ok(stage.backend("openai"), &graph).await;
assert_eq!(mock.calls_async().await, 1);
let work = work_stage(&state);
assert_eq!(work.response.as_deref(), Some("Summarized."));
assert_eq!(work.usage.tokens.input, INPUT_TOKENS_PER_CALL);
assert_eq!(work.usage.tokens.output, OUTPUT_TOKENS_PER_CALL);
assert_eq!(
work.usage.cost,
Some(Cost {
usd_micros: INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL,
source: CostSource::Catalog,
}),
"the response came priced from the catalog by lithos-llm's resolver"
);
let model = work.model.as_ref().expect("the stage names its model");
assert_eq!(model.provider.as_str(), PROVIDER);
assert_eq!(model.model_id.as_str(), MODEL);
assert!(
work.usage_by_model.is_empty(),
"a one-shot stage has one route; the split is the usage itself"
);
}

View file

@ -74,7 +74,7 @@ export interface RunStage {
*/
'started_at'?: string | null;
/**
* Usage for this stage execution alone. `cost` is the provider\'s reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/usage` rows use. All-zero counts mean the stage made no model calls. Unlike the usage rows, which sum every visit of a node, this covers only this visit.
* Usage for this stage execution alone. `cost` sums what lithos-llm attached to each answer: the provider\'s reported cost when there is one, otherwise the catalog\'s price for the route — the same figures the `/runs/{id}/usage` rows sum. All-zero counts mean the stage made no model calls. Unlike the usage rows, which sum every visit of a node, this covers only this visit.
*/
'usage': Usage;
}

View file

@ -101,7 +101,7 @@ export interface StageProjection {
'live_tool_ms'?: number;
'tool_batch'?: StageToolBatchProjection | null;
/**
* The stage\'s usage: while the stage runs, its agent\'s own accounting of the session tree with whatever cost the provider reported; once it ends, the same tokens with the catalog\'s price where the provider reported none.
* The stage\'s usage: the session tree\'s tokens and cost, live and once the stage ends. Every answer is priced once by lithos-llm (the provider\'s reported cost, else the catalog\'s price for the route) and summed; the cost is absent only when an answer had neither.
*/
'usage': Usage;
'model'?: UsageModelRef | null;