diff --git a/apps/fabro-web/app/lib/stage-sidebar.ts b/apps/fabro-web/app/lib/stage-sidebar.ts index b0b5508ae..b43c6b529 100644 --- a/apps/fabro-web/app/lib/stage-sidebar.ts +++ b/apps/fabro-web/app/lib/stage-sidebar.ts @@ -33,8 +33,9 @@ export interface Stage { startedAt: string | null; providerUsed: StageModelUsage | null; /** - * Tokens and cost for this visit alone, priced the same way the Usage tab - * prices its per-node rows. All-zero counts mean the stage called no model. + * Tokens and cost for this visit alone, the same figures the Usage tab + * sums into its per-node rows. All-zero counts mean the stage called no + * model. */ usage: Usage; } diff --git a/apps/fabro-web/app/routes/run-usage.tsx b/apps/fabro-web/app/routes/run-usage.tsx index 1e600b7e2..4d976e042 100644 --- a/apps/fabro-web/app/routes/run-usage.tsx +++ b/apps/fabro-web/app/routes/run-usage.tsx @@ -95,7 +95,7 @@ function TokenBreakdown({ usage }: { usage: Usage }) { ))}

- Includes subagent tokens, priced at each subagent's model. + Includes subagent tokens, each priced at the model it ran on.

); diff --git a/docs/internal/events.md b/docs/internal/events.md index 5ca1c44ec..e90e8dc51 100644 --- a/docs/internal/events.md +++ b/docs/internal/events.md @@ -412,7 +412,7 @@ Emitted when a workflow node finishes execution. | `suggested_next_ids` | string[] | Suggested successor node ids | | `usage` | object? | The stage's usage under the model it ran on (`ModelUsage`): for an agent stage, the whole session tree's tokens under the root's route. Absent for a stage that made no model calls | | `usage.model` | object | `provider`, `model_id`, and optional `speed` tier | -| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost is the provider's reported figure when it gave one, else the catalog's price; absent when the catalog has no rates | +| `usage.usage` | object | lithos-llm's `Usage`: `tokens` (the five disjoint buckets) and an optional `cost` (`usd_micros`, `source`). The cost sums what lithos-llm attached to each answer, the provider's reported figure when it gave one, else the catalog's price for the route; absent when an answer had neither | | `usage_by_model` | array? | For an agent stage, `usage` split by model: the root session's route and each subagent's own model, a subagent whose model the catalog does not know priced at the root's. Each row is a `ModelUsage`, and the rows sum to `usage`. Empty for stages without a coding agent and on events written before it existed | | `error` | string? | Error message (flattened from failure detail) | | `failure_class` | string? | `"transient_infra"`, `"deterministic"`, `"budget_exhausted"`, `"compilation_loop"`, `"canceled"`, `"structural"` | diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml index 29abc9088..e620dbd55 100644 --- a/docs/public/api-reference/fabro-api.yaml +++ b/docs/public/api-reference/fabro-api.yaml @@ -11195,10 +11195,11 @@ components: usage: $ref: "#/components/schemas/Usage" description: >- - The stage's usage: while the stage runs, its agent's own - accounting of the session tree with whatever cost the provider - reported; once it ends, the same tokens with the catalog's price - where the provider reported none. + The stage's usage: the session tree's tokens and cost, live and + once the stage ends. Every answer is priced once by lithos-llm + (the provider's reported cost, else the catalog's price for the + route) and summed; the cost is absent only when an answer had + neither. model: oneOf: - $ref: "#/components/schemas/UsageModelRef" @@ -13839,12 +13840,12 @@ components: usage: $ref: "#/components/schemas/Usage" description: >- - Usage for this stage execution alone. `cost` is the provider's - reported cost when there is one, otherwise the server catalog's - price for these tokens — the same pricing the `/runs/{id}/usage` - rows use. All-zero counts mean the stage made no model calls. - Unlike the usage rows, which sum every visit of a node, this - covers only this visit. + Usage for this stage execution alone. `cost` sums what lithos-llm + attached to each answer: the provider's reported cost when there + is one, otherwise the catalog's price for the route — the same + figures the `/runs/{id}/usage` rows sum. All-zero counts mean the + stage made no model calls. Unlike the usage rows, which sum every + visit of a node, this covers only this visit. # ── File Diff Schemas ────────────────────────────────────────────── diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs index b5dd8876e..04b838b95 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs @@ -461,12 +461,12 @@ mod tests { use fabro_workflow::event::{ Event, RunNoticeLevel, SandboxLifecycle, to_run_event, to_run_event_at, }; - use fabro_workflow::outcome::model_usage_from_llm; + use fabro_workflow::outcome::ModelUsage; use lithos_llm::catalog::{ModelId, builtin}; - use lithos_llm::types::TokenCounts; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use pebble_coding_agent::events::{ CodingAgentEvent, CodingEvent, CompactionReason, ErrorData as AgentErrorData, - ErrorKind as AgentErrorKind, Usage, + ErrorKind as AgentErrorKind, }; use super::*; @@ -649,18 +649,22 @@ mod tests { preferred_label: None, suggested_next_ids: Vec::new(), usage_by_model: Vec::new(), - usage: Some( - model_usage_from_llm( - &fabro_llm::test_support::test_catalog(), - &ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")), - TokenCounts { + // Priced as lithos-llm prices gpt-5.4: 1200 input at $2.50/M and + // 300 output at $15/M. + usage: Some(ModelUsage::new( + ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")), + Usage { + tokens: TokenCounts { input: 1200, output: 300, ..TokenCounts::default() }, - ) - .unwrap(), - ), + cost: Some(Cost { + usd_micros: 7_500, + source: CostSource::Catalog, + }), + }, + )), failure: None, notes: None, files_touched: Vec::new(), diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs index 6f3868109..ab533e1da 100644 --- a/lib/components/fabro-store/src/run_state.rs +++ b/lib/components/fabro-store/src/run_state.rs @@ -716,8 +716,8 @@ fn apply_agent_event( // Pebble's own fold sees every agent event the stage stored, before the // fabro-only arms below read the same event. While the stage runs, its // usage is that fold's: the tree's tokens, the root's and every - // subagent's, with whatever cost the provider reported. The terminal - // usage then brings the catalog's price for the same tokens. + // subagent's, with the cost lithos-llm attached to each answer. The + // terminal usage is the same sum, split by model. if let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) { let agent = stage.agent.get_or_insert_default(); agent.apply(&props.event); @@ -5342,11 +5342,105 @@ mod tests { } /// One usage rule: a stage's usage is its session tree's, live and at - /// completion. The terminal usage carries the tokens the fold already - /// showed plus the catalog's price, so completion changes the cost, not - /// the tokens, and keeps the split by model. + /// completion, cost included. lithos-llm prices each answer once, pebble + /// sums them, and the terminal usage is the same sum, so completion + /// changes neither the tokens nor the cost; it adds the split by model. #[test] - fn stage_completed_keeps_the_trees_live_usage_and_prices_it() { + fn stage_completed_keeps_the_trees_live_usage_and_its_cost() { + let mut state = initialized_projection(); + let stage_id = StageId::new("build", 1); + let model = priced_usage().model().clone(); + let catalog_priced = |input: u64, output: u64, usd_micros: u64| Usage { + tokens: TokenCounts { + input, + output, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), + }; + let message = |session: &str, usage: Usage| { + let mut event = CodingAgentEvent::new( + session, + CodingEvent::AssistantMessage { + text: "assistant text".to_string(), + model: model.model_id.to_string(), + usage, + tool_call_count: 0, + context_window: None, + reasoning: None, + }, + SystemTime::UNIX_EPOCH, + ); + if session != "ses_test" { + event = event.with_parent_session_id("ses_test"); + } + EventBody::Agent(AgentEventProps::new("code", 1, event)) + }; + + state + .apply_event(&test_stage_event( + 1, + EventBody::StageStarted(started_props()), + stage_id.clone(), + )) + .unwrap(); + state + .apply_event(&test_stage_event( + 2, + activated(model.provider.as_str(), model.model_id.as_str()), + stage_id.clone(), + )) + .unwrap(); + state + .apply_event(&test_stage_event( + 3, + message("ses_test", catalog_priced(100, 50, 300)), + stage_id.clone(), + )) + .unwrap(); + state + .apply_event(&test_stage_event( + 4, + message("ses_child", catalog_priced(7, 1, 21)), + stage_id.clone(), + )) + .unwrap(); + let live = state.stage(&stage_id).unwrap().usage; + assert_eq!( + live, + catalog_priced(107, 51, 321), + "the subagent's tokens and cost are the stage's too" + ); + + // The terminal usage is the same sum, under the root's route. + let tree = ModelUsage::new(model.clone(), live); + let mut props = completed_props(42, StageOutcome::Succeeded); + props.usage = Some(tree.clone()); + props.usage_by_model = vec![tree.clone()]; + state + .apply_event(&test_stage_event( + 5, + EventBody::StageCompleted(props), + stage_id.clone(), + )) + .unwrap(); + + let stage = state.stage(&stage_id).unwrap(); + assert_eq!( + stage.usage, live, + "completion keeps the usage the fold showed, cost included" + ); + assert_eq!(stage.model.as_ref(), Some(&model)); + assert_eq!(stage.usage_by_model, vec![tree]); + } + + /// An answer nobody priced leaves the tree's cost unknown, live and at + /// completion alike; the tokens are still counted. + #[test] + fn an_unpriced_answer_leaves_the_stage_cost_unknown_live_and_at_completion() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); let model = priced_usage().model().clone(); @@ -5372,57 +5466,28 @@ mod tests { stage_id.clone(), )) .unwrap(); + let live = state.stage(&stage_id).unwrap().usage; + assert_eq!(live, live_counts(100, 50)); + assert_eq!(live.cost, None); + + let tree = ModelUsage::new(model.clone(), live); + let mut props = completed_props(42, StageOutcome::Succeeded); + props.usage = Some(tree.clone()); + props.usage_by_model = vec![tree]; state .apply_event(&test_stage_event( 4, - child_message_body(7, 1), - stage_id.clone(), - )) - .unwrap(); - let live = state.stage(&stage_id).unwrap().usage; - assert_eq!( - live, - live_counts(107, 51), - "the subagent's tokens are the stage's too" - ); - - let tree = ModelUsage::new(model.clone(), Usage { - tokens: TokenCounts { - input: 107, - output: 51, - ..TokenCounts::default() - }, - cost: Some(Cost { - usd_micros: 321, - source: CostSource::Catalog, - }), - }); - let mut props = completed_props(42, StageOutcome::Succeeded); - props.usage = Some(tree.clone()); - props.usage_by_model = vec![tree.clone()]; - state - .apply_event(&test_stage_event( - 5, EventBody::StageCompleted(props), stage_id.clone(), )) .unwrap(); let stage = state.stage(&stage_id).unwrap(); + assert_eq!(stage.usage, live); assert_eq!( - stage.usage.tokens, live.tokens, - "completion keeps the tokens the fold showed" + stage.usage.cost, None, + "nothing priced it, so nothing invents a cost" ); - assert_eq!( - stage.usage.cost, - Some(Cost { - usd_micros: 321, - source: CostSource::Catalog, - }), - "and brings the catalog's price" - ); - assert_eq!(stage.model.as_ref(), Some(&model)); - assert_eq!(stage.usage_by_model, vec![tree]); } #[test] diff --git a/lib/components/fabro-workflow/src/handler/llm/pebble.rs b/lib/components/fabro-workflow/src/handler/llm/pebble.rs index 65b7b143e..2ac99daa3 100644 --- a/lib/components/fabro-workflow/src/handler/llm/pebble.rs +++ b/lib/components/fabro-workflow/src/handler/llm/pebble.rs @@ -63,7 +63,7 @@ use crate::context::keys::Fidelity; use crate::error::Error; use crate::event::{Emitter, Event, StageScope}; use crate::model_fallback::{ModelFallbackNotice, ModelFallbackPolicy}; -use crate::outcome::{Outcome, model_usage_from_llm, with_reported_cost}; +use crate::outcome::Outcome; use crate::services::FabroRunToolServices; use crate::steering_hub::SteeringHub; use crate::web_search::{self, SearchSecrets}; @@ -336,20 +336,18 @@ struct StageUsage { by_model: Vec, } -/// Prices the stage's account from the catalog: the root session at -/// `root_model`, its route, and each descendant at its own route where the -/// catalog knows it and at the root's otherwise, so a subagent on a cheaper -/// or dearer model is priced as what it ran. A descendant on the root's -/// route joins the root's row. Where pebble carried a provider-reported -/// cost, that cost stands in for the catalog's estimate. The total's cost is -/// the rows' sum, which is `None` once a row that used tokens has no cost. +/// The stage's account grouped by model: the root session at `root_model`, +/// its route, and each descendant at its own route where the catalog knows +/// it and at the root's otherwise. A descendant on the root's route joins +/// the root's row. Every cost is the one pebble carried: lithos-llm attaches +/// the provider's reported cost or the catalog's price to each answer, and +/// pebble sums them per session, so fabro prices nothing of its own. A row, +/// and the total, has a cost only when every answer in it was priced. fn stage_usage( catalog: &Catalog, root_model: &ModelRef, account: &SessionProjection, -) -> Result { - // Each group's usage is the sum of pebble's accounts, so its cost is what - // the provider reported, or `None` once an unpriced account is in it. +) -> StageUsage { let mut groups: Vec<(ModelRef, Usage)> = vec![(root_model.clone(), account.usage)]; for descendant in account.descendants.values() { let model = descendant_model(catalog, root_model, descendant); @@ -361,25 +359,20 @@ fn stage_usage( // The root's row first, then the others by model. groups[1..].sort_by(|left, right| left.0.sort_key().cmp(&right.0.sort_key())); - let mut by_model = Vec::with_capacity(groups.len()); - let mut total = Usage::default(); - for (model, usage) in groups { - let row = with_reported_cost( - model_usage_from_llm(catalog, &model, usage.tokens)?, - usage.cost, - ); - total = total.saturating_add(row.usage); - by_model.push(row); + let total = fabro_types::sum_usage(groups.iter().map(|(_, usage)| *usage)); + StageUsage { + total: ModelUsage::new(root_model.clone(), total), + by_model: groups + .into_iter() + .map(|(model, usage)| ModelUsage::new(model, usage)) + .collect(), } - Ok(StageUsage { - total: ModelUsage::new(root_model.clone(), total), - by_model, - }) } -/// The route a descendant is billed at: its own where its start named one -/// the catalog knows, else the root's. A descendant whose start was not seen -/// names only its answers' model, taken to be on the root's provider. +/// The route a descendant's usage is grouped under: its own where its start +/// named one the catalog knows, else the root's. A descendant whose start +/// was not seen names only its answers' model, taken to be on the root's +/// provider. fn descendant_model( catalog: &Catalog, root_model: &ModelRef, @@ -756,27 +749,17 @@ impl PebbleBackend { /// The failed outcome of an agent stage that spent before it failed: the /// failure itself, with the session tree's usage, the files it wrote, and - /// its active time, so the run records what the stage spent. A usage the - /// catalog cannot price is logged and left off. + /// its active time, so the run records what the stage spent. fn failed_outcome(&self, error: &Error, live: &LiveAgent, plan: &FallbackPlan) -> Outcome { let mut outcome = error.to_fail_outcome(); let account = live.account(); - match stage_usage( + let usage = stage_usage( self.catalog.as_ref(), &route_model(plan.current()), &account, - ) { - Ok(usage) => { - outcome.usage = Some(usage.total); - outcome.usage_by_model = usage.by_model; - } - Err(usage_error) => { - tracing::debug!( - error = %usage_error, - "failed agent stage could not be priced" - ); - } - } + ); + outcome.usage = Some(usage.total); + outcome.usage_by_model = usage.by_model; outcome.files_touched = account.files_touched; outcome.timing = Some(StageTiming::active_only( crate::millis_u64(live.inference_duration), @@ -1028,12 +1011,10 @@ impl CodergenBackend for PebbleBackend { continue; } - // The provider's own cost, when every answer carried one, stands in - // for the catalog's estimate. - let stage_usage = with_reported_cost( - model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage.tokens)?, - total_usage.cost, - ); + // Each response came priced by lithos-llm: the provider's reported + // cost, or the catalog's price for the route. The stage's cost is + // their sum, known only when every answer was priced. + let stage_usage = ModelUsage::new(completion.model.clone(), total_usage); return Ok(CodergenResult::Text { text: response_text, @@ -1238,7 +1219,7 @@ impl CodergenBackend for PebbleBackend { self.catalog.as_ref(), &route_model(fallback_plan.current()), &account, - )?; + ); live.release_lease(); match reuse_key { @@ -1303,7 +1284,7 @@ mod tests { } } - fn message(model: &str, input: u64, output: u64, cost: Option) -> CodingEvent { + fn message(model: &str, input: u64, output: u64, cost: Option) -> CodingEvent { CodingEvent::AssistantMessage { text: "ok".to_string(), model: model.to_string(), @@ -1313,10 +1294,7 @@ mod tests { output, ..TokenCounts::default() }, - cost: cost.map(|usd_micros| Cost { - usd_micros, - source: CostSource::Provider, - }), + cost, }, tool_call_count: 0, context_window: None, @@ -1324,6 +1302,13 @@ mod tests { } } + fn catalog_cost(usd_micros: u64) -> Cost { + Cost { + usd_micros, + source: CostSource::Catalog, + } + } + fn root_model() -> ModelRef { ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")) } @@ -1334,8 +1319,10 @@ mod tests { account } + /// Every cost comes from pebble's stream, where lithos-llm attached it + /// to each answer; fabro groups and sums, and prices nothing itself. #[test] - fn stage_usage_prices_the_root_at_its_route_and_each_descendant_at_its_own() { + fn stage_usage_groups_pebbles_priced_accounts_by_route_and_sums_them() { let catalog = test_catalog(); let account = account(&[ root(started("openai", "gpt-5.4")), @@ -1344,20 +1331,43 @@ mod tests { content: None, source: InputSource::Prompt, }), - root(message("gpt-5.4", 100_000, 25_000, None)), + root(message( + "gpt-5.4", + 100_000, + 25_000, + Some(catalog_cost(300_000)), + )), // A child on the parent's route joins the parent's row. child("ses_same", started("openai", "gpt-5.4")), - child("ses_same", message("gpt-5.4", 10_000, 1_000, None)), - // A child on another route is its own row, at that route's rate. + child( + "ses_same", + message("gpt-5.4", 10_000, 1_000, Some(catalog_cost(30_000))), + ), + // A child on another route is its own row, at the cost its + // provider reported. child("ses_other", started("anthropic", "claude-sonnet-5")), - child("ses_other", message("claude-sonnet-5", 20_000, 2_000, None)), - // A child on a route the catalog does not know bills at the root's. + child( + "ses_other", + message( + "claude-sonnet-5", + 20_000, + 2_000, + Some(Cost { + usd_micros: 70_000, + source: CostSource::Provider, + }), + ), + ), + // A child on a route the catalog does not know joins the root's row. child("ses_unknown", started("nowhere", "mystery")), - child("ses_unknown", message("mystery", 1_000, 100, None)), + child( + "ses_unknown", + message("mystery", 1_000, 100, Some(catalog_cost(5_000))), + ), root(CodingEvent::ProcessingEnd), ]); - let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); + let usage = stage_usage(&catalog, &root_model(), &account); assert_eq!(usage.by_model.len(), 2, "{:?}", usage.by_model); let root_row = &usage.by_model[0]; @@ -1367,102 +1377,103 @@ mod tests { "the root, the same-route child, and the unknown-route child" ); assert_eq!(root_row.usage.tokens.output, 26_100); - let root_priced = - model_usage_from_llm(&catalog, &root_model(), root_row.usage.tokens).unwrap(); - assert_eq!(root_row.usage.cost, root_priced.usage.cost); assert_eq!( - root_row.usage.cost.map(|cost| cost.source), - Some(CostSource::Catalog) + root_row.usage.cost, + Some(catalog_cost(335_000)), + "the row's cost is the sum of what pebble carried, still the catalog's" ); - let other_model = ModelRef::new( - ProviderId::new("anthropic"), - ModelId::new("claude-sonnet-5"), - ); let other_row = &usage.by_model[1]; - assert_eq!(other_row.model, other_model); + assert_eq!( + other_row.model, + ModelRef::new( + ProviderId::new("anthropic"), + ModelId::new("claude-sonnet-5"), + ) + ); assert_eq!(other_row.usage.tokens.input, 20_000); - assert_eq!(other_row.usage.tokens.output, 2_000); - let other_priced = - model_usage_from_llm(&catalog, &other_model, other_row.usage.tokens).unwrap(); - assert_eq!(other_row.usage.cost, other_priced.usage.cost); - assert_ne!( + assert_eq!( other_row.usage.cost, - model_usage_from_llm(&catalog, &root_model(), other_row.usage.tokens) - .unwrap() - .usage - .cost, - "priced at its own rate, not the root's" + Some(Cost { + usd_micros: 70_000, + source: CostSource::Provider, + }), + "a provider-reported cost is kept as reported" ); - // The total is the tree's tokens under the root's route, at the rows' summed - // cost, from the catalog like every row. + // The total is the tree's tokens under the root's route; its cost is + // the rows' sum, assembled from two sources. assert_eq!(usage.total.model, root_model()); assert_eq!(usage.total.usage.tokens.input, 131_000); assert_eq!(usage.total.usage.tokens.output, 28_100); assert_eq!( usage.total.usage.cost, Some(Cost { - usd_micros: root_priced.usage.cost.unwrap().usd_micros - + other_priced.usage.cost.unwrap().usd_micros, - source: CostSource::Catalog, - }) - ); - } - - #[test] - fn a_provider_reported_cost_stands_in_for_the_catalogs_estimate() { - let catalog = test_catalog(); - let account = account(&[ - root(started("openai", "gpt-5.4")), - root(message("gpt-5.4", 1_000, 100, Some(4_321))), - child("ses_child", started("anthropic", "claude-sonnet-5")), - child("ses_child", message("claude-sonnet-5", 500, 50, None)), - ]); - - let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); - - assert_eq!( - usage.by_model[0].usage.cost, - Some(Cost { - usd_micros: 4_321, - source: CostSource::Provider, - }) - ); - let child_priced = model_usage_from_llm( - &catalog, - &usage.by_model[1].model, - usage.by_model[1].usage.tokens, - ) - .unwrap(); - assert_eq!(usage.by_model[1].usage.cost, child_priced.usage.cost); - // One row reported, one priced: the sum is the application's. - assert_eq!( - usage.total.usage.cost, - Some(Cost { - usd_micros: 4_321 + child_priced.usage.cost.unwrap().usd_micros, + usd_micros: 405_000, source: CostSource::Application, }) ); } + /// An answer pebble could not price (a model with no catalog price and + /// no provider cost) leaves its row's cost, and the total's, unknown; the + /// tokens are still counted. Live and completed usage agree because both + /// are the same sum of pebble's accounts. #[test] - fn a_descendant_seen_only_through_its_answers_bills_on_the_roots_provider() { + fn stage_usage_leaves_the_cost_unknown_once_an_answer_was_unpriced() { + let catalog = test_catalog(); + let priced_only = account(&[ + root(started("openai", "gpt-5.4")), + root(message("gpt-5.4", 1_000, 100, Some(catalog_cost(4_321)))), + ]); + let priced = stage_usage(&catalog, &root_model(), &priced_only); + assert_eq!(priced.total.usage.cost, Some(catalog_cost(4_321))); + assert_eq!( + priced.total.usage, + priced_only + .usage + .saturating_add(priced_only.descendant_usage()), + "the completed usage is the live fold's, cost included" + ); + + let tree = account(&[ + root(started("openai", "gpt-5.4")), + root(message("gpt-5.4", 1_000, 100, Some(catalog_cost(4_321)))), + child("ses_child", started("anthropic", "claude-sonnet-5")), + child("ses_child", message("claude-sonnet-5", 500, 50, None)), + ]); + + let usage = stage_usage(&catalog, &root_model(), &tree); + + assert_eq!(usage.by_model[0].usage.cost, Some(catalog_cost(4_321))); + assert_eq!(usage.by_model[1].usage.tokens.input, 500); + assert_eq!(usage.by_model[1].usage.cost, None); + assert_eq!(usage.total.usage.tokens.input, 1_500); + assert_eq!(usage.total.usage.cost, None); + assert_eq!( + usage.total.usage, + tree.usage.saturating_add(tree.descendant_usage()), + "the completed usage is the live fold's, cost unknown at both" + ); + } + + #[test] + fn a_descendant_seen_only_through_its_answers_groups_under_the_roots_provider() { let catalog = test_catalog(); let mut account = account(&[root(started("openai", "gpt-5.4"))]); // No `SessionStarted` for the child: only its answer names a model. account.apply(&child( "ses_quiet", - message("gpt-5.4-mini", 1_000, 100, None), + message("gpt-5.4-mini", 1_000, 100, Some(catalog_cost(1))), )); - let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); + let usage = stage_usage(&catalog, &root_model(), &account); let child_row = usage .by_model .iter() .find(|row| row.model.model_id.as_str() == "gpt-5.4-mini") - .expect("the child is billed as its answers' model on the root's provider"); + .expect("the child is grouped as its answers' model on the root's provider"); assert_eq!(child_row.model.provider, root_model().provider); assert_eq!(child_row.usage.tokens.input, 1_000); } diff --git a/lib/components/fabro-workflow/src/handler/llm/preamble.rs b/lib/components/fabro-workflow/src/handler/llm/preamble.rs index 1fade571b..68e1bcf58 100644 --- a/lib/components/fabro-workflow/src/handler/llm/preamble.rs +++ b/lib/components/fabro-workflow/src/handler/llm/preamble.rs @@ -591,22 +591,20 @@ mod tests { use fabro_graphviz::graph::AttrValue; use fabro_types::ModelRef; use lithos_llm::catalog::{ModelId, builtin}; - use lithos_llm::types::TokenCounts; + use lithos_llm::types::{TokenCounts, Usage}; use super::*; - use crate::outcome::{ModelUsage, model_usage_from_llm}; + use crate::outcome::ModelUsage; fn stage_usage(model: &str, input: u64, output: u64) -> ModelUsage { - model_usage_from_llm( - &fabro_llm::test_support::test_catalog(), - &ModelRef::new(builtin::anthropic(), ModelId::new(model)), - TokenCounts { + ModelUsage::new( + ModelRef::new(builtin::anthropic(), ModelId::new(model)), + Usage::from(TokenCounts { input, output, ..TokenCounts::default() - }, + }), ) - .unwrap() } fn large_prompt_value(bytes: u64, path: &str, preview: &str) -> serde_json::Value { diff --git a/lib/components/fabro-workflow/src/outcome.rs b/lib/components/fabro-workflow/src/outcome.rs index 0a1d0fc22..45f69752b 100644 --- a/lib/components/fabro-workflow/src/outcome.rs +++ b/lib/components/fabro-workflow/src/outcome.rs @@ -1,46 +1,12 @@ pub use fabro_core::outcome::{ FailureCategory, FailureDetail, OutcomeMeta, StageOutcome, StageState, }; -use fabro_llm::lithos_catalog::Catalog; -use fabro_types::ModelRef; pub use fabro_types::ModelUsage; -use lithos_llm::types::{Cost, TokenCounts, Usage}; -use crate::error::{Error, FailureSignature, classify_failure_reason}; +use crate::error::{FailureSignature, classify_failure_reason}; pub type Outcome = fabro_core::Outcome>; -/// Prices `tokens` on `model` from the catalog: the usage carries a -/// [`CostSource::Catalog`](lithos_llm::types::CostSource::Catalog) cost when -/// the catalog has rates for the model, and no cost otherwise. -/// -/// The provider must be one the catalog knows; a passthrough model on a known -/// provider is priced with no cost, since the catalog has no rates for it. -pub fn model_usage_from_llm( - catalog: &Catalog, - model: &ModelRef, - tokens: TokenCounts, -) -> Result { - if catalog.enabled_provider(model.provider.as_str()).is_none() { - return Err(Error::Precondition(format!( - "Provider \"{}\" is not configured", - model.provider - ))); - } - let cost = catalog.estimate_cost(&model.handle(), tokens, model.speed); - Ok(ModelUsage::new(model.clone(), Usage { tokens, cost })) -} - -/// `usage` with `cost` in place of whatever it carried, when a provider -/// reported one; `None` keeps the usage as it is. -#[must_use] -pub fn with_reported_cost(mut usage: ModelUsage, cost: Option) -> ModelUsage { - if let Some(cost) = cost { - usage.usage.cost = Some(cost); - } - usage -} - pub trait OutcomeExt: Sized { fn fail_deterministic(reason: impl Into) -> Self; fn fail_classify(reason: impl Into) -> Self; @@ -131,75 +97,7 @@ pub fn format_cost(cost: f64) -> String { #[cfg(test)] mod tests { - use fabro_llm::lithos_catalog::Catalog; - use fabro_llm::test_support::{test_catalog, test_catalog_with_overlay}; - use fabro_types::ModelRef; - use lithos_llm::catalog::{ModelId, ProviderId, builtin}; - use lithos_llm::types::{Cost, CostSource, Speed, TokenCounts}; - - use super::{OutcomeExt, model_usage_from_llm, with_reported_cost}; - - fn model_ref(provider: ProviderId, model_id: &str, speed: Option) -> ModelRef { - ModelRef::new(provider, ModelId::new(model_id)).with_speed(speed) - } - - fn catalog() -> Catalog { - test_catalog() - } - - #[test] - fn model_usage_from_llm_prices_openai_cached_input_and_reasoning_output() { - // Stay under the 272k long-context tier so the standard rates apply. - let usage = TokenCounts { - input: 100_000, - output: 25_000, - reasoning: 5_000, - cache_read: 50_000, - ..TokenCounts::default() - }; - let billed = model_usage_from_llm( - &catalog(), - &model_ref(builtin::openai(), "gpt-5.4", None), - usage, - ) - .unwrap(); - - // 100k input at $2.50/M + 50k cached at $0.25/M + 30k output at $15/M. - assert_eq!( - billed.usage.cost, - Some(Cost { - usd_micros: 712_500, - source: CostSource::Catalog, - }) - ); - assert_eq!(billed.usage.tokens.output, 25_000); - assert_eq!(billed.usage.tokens.reasoning, 5_000); - } - - #[test] - fn response_cost_overrides_catalog_estimate() { - let usage = TokenCounts { - input: 11, - output: 7, - ..TokenCounts::default() - }; - let reported = Cost { - usd_micros: 125_000, - source: CostSource::Provider, - }; - let billed = with_reported_cost( - model_usage_from_llm( - &catalog(), - &model_ref(builtin::openai(), "gpt-5.4", None), - usage, - ) - .unwrap(), - Some(reported), - ); - - assert_eq!(billed.usage.cost, Some(reported)); - assert_eq!(billed.usage.tokens, usage); - } + use super::OutcomeExt; #[test] fn retry_classify_marks_failed_outcome_with_retry_request() { @@ -210,94 +108,4 @@ mod tests { }); assert!(outcome.status.retry_requested()); } - - #[test] - fn model_usage_from_llm_prices_anthropic_fast_mode_cache_write_rates() { - let usage = TokenCounts { - input: 100_000, - output: 10_000, - reasoning: 5_000, - cache_read: 20_000, - cache_write: 30_000, - }; - let billed = model_usage_from_llm( - &catalog(), - &model_ref(builtin::anthropic(), "claude-opus-5", Some(Speed::Fast)), - usage, - ) - .unwrap(); - - // Fast rates: $10/M input, $50/M output (incl. reasoning), $1/M cache - // read, $12.50/M cache write. - assert_eq!( - billed.usage.cost.map(|cost| cost.usd_micros), - Some(2_145_000) - ); - } - - #[test] - fn model_usage_from_llm_uses_injected_custom_catalog() { - let catalog = test_catalog_with_overlay( - r#" -[providers.proxy] -display_name = "Proxy" -adapter = "openai-compatible" -codec = "openai-chat" -base_url = "https://proxy.example/v1" -auth = { type = "bearer" } -default_model = "canonical-model" - -[providers.proxy.models.canonical-model] -display_name = "Canonical Model" -api_model = "wire-model" -limits = { context_tokens = 1000, max_output_tokens = 500 } -capabilities = { text = true, tools = true } -pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_million = 2000000 } -"#, - ); - let usage = TokenCounts { - input: 1_000_000, - output: 500_000, - ..TokenCounts::default() - }; - let billed = model_usage_from_llm( - &catalog, - &model_ref(ProviderId::new("proxy"), "canonical-model", None), - usage, - ) - .unwrap(); - - assert_eq!( - billed.usage.cost.map(|cost| cost.usd_micros), - Some(2_000_000) - ); - assert_eq!(billed.model_id(), "canonical-model"); - } - - #[test] - fn passthrough_model_on_known_provider_has_no_cost() { - let billed = model_usage_from_llm( - &catalog(), - &model_ref(builtin::openai(), "brand-new-model", None), - TokenCounts { - input: 10, - output: 5, - ..TokenCounts::default() - }, - ) - .unwrap(); - assert_eq!(billed.usage.cost, None); - assert_eq!(billed.usage.tokens.input, 10); - } - - #[test] - fn unknown_provider_is_a_precondition_failure() { - let error = model_usage_from_llm( - &catalog(), - &model_ref(ProviderId::new("nowhere"), "model", None), - TokenCounts::default(), - ) - .unwrap_err(); - assert!(error.to_string().contains("not configured"), "{error}"); - } } diff --git a/lib/components/fabro-workflow/tests/it/pebble_agent.rs b/lib/components/fabro-workflow/tests/it/pebble_agent.rs index 1842d9e3d..8f54f5dbf 100644 --- a/lib/components/fabro-workflow/tests/it/pebble_agent.rs +++ b/lib/components/fabro-workflow/tests/it/pebble_agent.rs @@ -44,6 +44,7 @@ use fabro_workflow::test_support::WorkflowRunner; use httpmock::Method::POST; use httpmock::MockServer; use lithos_llm::catalog::ProviderId; +use lithos_llm::types::{Cost, CostSource}; use pebble_coding_agent::events::{CodingEvent, FailoverContinuation, FailoverStop}; use tokio_util::sync::CancellationToken; @@ -235,6 +236,19 @@ fn agent_registry(backend: PebbleBackend) -> HandlerRegistry { registry } +fn prompt_registry(backend: PebbleBackend) -> HandlerRegistry { + let mut registry = HandlerRegistry::new(Box::new(StartHandler)); + registry.register("start", Box::new(StartHandler)); + registry.register("exit", Box::new(ExitHandler)); + registry.register( + "prompt", + Box::new(fabro_workflow::handler::prompt::PromptHandler::new(Some( + Box::new(backend), + ))), + ); + registry +} + /// Every run event the run emitted, in order. type Events = Arc>>; @@ -339,6 +353,25 @@ impl Stage { state } + /// Runs `graph` with `work` as a one-shot prompt stage and returns the + /// projection. + async fn run_prompt_ok( + &self, + backend: PebbleBackend, + graph: &Graph, + ) -> fabro_types::RunProjection { + let sandbox = local_sandbox(self.dir.path()).await; + let runner = + WorkflowRunner::new(prompt_registry(backend), Arc::clone(&self.emitter), sandbox); + let options = run_options(self.dir.path(), CancellationToken::new()); + let (outcome, state) = runner + .run_with_state(graph, &options) + .await + .expect("workflow execution should complete"); + assert_eq!(outcome.status, StageOutcome::Succeeded, "{outcome:?}"); + state + } + /// Fires `action` once, when the stage's first model call starts. fn on_first_llm_call(&self, action: impl Fn() + Send + Sync + 'static) { let fired = AtomicBool::new(false); @@ -408,7 +441,7 @@ async fn write_file_under_profile(profile: &str, tool: &str, path_key: &str) { assert_eq!( work.usage.cost.map(|cost| cost.usd_micros), Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), - "{profile}: cost from the catalog's pricing" + "{profile}: every answer came priced from the catalog" ); let checkpoint = state.current_checkpoint().expect("a checkpoint"); let outcome = checkpoint @@ -831,7 +864,7 @@ async fn a_stage_that_fails_after_spending_bills_what_it_spent() { assert_eq!( work.usage.cost.map(|cost| cost.usd_micros), Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), - "priced from the catalog like a completed stage" + "every answer came priced from the catalog" ); assert_eq!(work.usage_by_model.len(), 1, "{:?}", work.usage_by_model); assert_eq!( @@ -1006,7 +1039,7 @@ async fn a_subagent_runs_under_its_parent_session() { assert_eq!( work.usage.cost.map(|cost| cost.usd_micros), Some(4 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), - "priced from the catalog for every call" + "every answer came priced from the catalog" ); let agent = work .agent @@ -1014,9 +1047,14 @@ async fn a_subagent_runs_under_its_parent_session() { .expect("the stage carries pebble's fold"); let descendants = agent.descendant_usage(); assert_eq!( - work.usage.tokens.input, - agent.usage.tokens.input + descendants.tokens.input, - "the completed usage is what the live fold showed" + work.usage, + agent.usage.saturating_add(descendants), + "the completed usage is what the live fold showed, cost included" + ); + assert_eq!( + work.usage.cost.map(|cost| cost.source), + Some(CostSource::Catalog), + "lithos-llm priced every answer from the catalog; fabro priced nothing" ); assert_eq!(descendants.tokens.input, INPUT_TOKENS_PER_CALL); // The child ran on its parent's model, so the split is one row carrying @@ -1613,3 +1651,70 @@ async fn daytona_sandbox_runs_an_agent_stage() { sandbox.delete().await.expect("Daytona cleanup failed"); } + +// --- One-shot prompt stages -------------------------------------------------- + +/// A one-shot prompt stage calls lithos-llm's client directly, and the +/// response comes back priced: the resolver fills the catalog's price for +/// the route when the provider reported none. Fabro records that cost as +/// is; it estimates nothing itself. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn a_prompt_stage_records_the_catalog_cost_lithos_attached_to_the_response() { + let stage = Stage::new().await; + let completion = serde_json::json!({ + "id": "chatcmpl-prompt", + "object": "chat.completion", + "model": MODEL, + "choices": [{ + "index": 0, + "message": { "role": "assistant", "content": "Summarized." }, + "finish_reason": "stop" + }], + "usage": { + "prompt_tokens": INPUT_TOKENS_PER_CALL, + "completion_tokens": OUTPUT_TOKENS_PER_CALL, + "total_tokens": INPUT_TOKENS_PER_CALL + OUTPUT_TOKENS_PER_CALL, + } + }); + let mock = stage + .server + .mock_async(|when, then| { + when.method(POST) + .path(CHAT_PATH) + .body_includes("Summarize the change"); + then.status(200) + .header("content-type", "application/json") + .body(completion.to_string()); + }) + .await; + + let mut graph = agent_graph("Prompt", "Summarize the change"); + graph + .nodes + .get_mut("work") + .expect("the work node") + .attrs + .insert("type".to_string(), AttrValue::String("prompt".to_string())); + let state = stage.run_prompt_ok(stage.backend("openai"), &graph).await; + + assert_eq!(mock.calls_async().await, 1); + let work = work_stage(&state); + assert_eq!(work.response.as_deref(), Some("Summarized.")); + assert_eq!(work.usage.tokens.input, INPUT_TOKENS_PER_CALL); + assert_eq!(work.usage.tokens.output, OUTPUT_TOKENS_PER_CALL); + assert_eq!( + work.usage.cost, + Some(Cost { + usd_micros: INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL, + source: CostSource::Catalog, + }), + "the response came priced from the catalog by lithos-llm's resolver" + ); + let model = work.model.as_ref().expect("the stage names its model"); + assert_eq!(model.provider.as_str(), PROVIDER); + assert_eq!(model.model_id.as_str(), MODEL); + assert!( + work.usage_by_model.is_empty(), + "a one-shot stage has one route; the split is the usage itself" + ); +} diff --git a/lib/packages/fabro-api-client/src/models/run-stage.ts b/lib/packages/fabro-api-client/src/models/run-stage.ts index d62351136..68f5591cd 100644 --- a/lib/packages/fabro-api-client/src/models/run-stage.ts +++ b/lib/packages/fabro-api-client/src/models/run-stage.ts @@ -74,7 +74,7 @@ export interface RunStage { */ 'started_at'?: string | null; /** - * Usage for this stage execution alone. `cost` is the provider\'s reported cost when there is one, otherwise the server catalog\'s price for these tokens — the same pricing the `/runs/{id}/usage` rows use. All-zero counts mean the stage made no model calls. Unlike the usage rows, which sum every visit of a node, this covers only this visit. + * Usage for this stage execution alone. `cost` sums what lithos-llm attached to each answer: the provider\'s reported cost when there is one, otherwise the catalog\'s price for the route — the same figures the `/runs/{id}/usage` rows sum. All-zero counts mean the stage made no model calls. Unlike the usage rows, which sum every visit of a node, this covers only this visit. */ 'usage': Usage; } diff --git a/lib/packages/fabro-api-client/src/models/stage-projection.ts b/lib/packages/fabro-api-client/src/models/stage-projection.ts index 9fcd9169a..0375483ce 100644 --- a/lib/packages/fabro-api-client/src/models/stage-projection.ts +++ b/lib/packages/fabro-api-client/src/models/stage-projection.ts @@ -101,7 +101,7 @@ export interface StageProjection { 'live_tool_ms'?: number; 'tool_batch'?: StageToolBatchProjection | null; /** - * The stage\'s usage: while the stage runs, its agent\'s own accounting of the session tree with whatever cost the provider reported; once it ends, the same tokens with the catalog\'s price where the provider reported none. + * The stage\'s usage: the session tree\'s tokens and cost, live and once the stage ends. Every answer is priced once by lithos-llm (the provider\'s reported cost, else the catalog\'s price for the route) and summed; the cost is absent only when an answer had neither. */ 'usage': Usage; 'model'?: UsageModelRef | null;