diff --git a/Cargo.lock b/Cargo.lock index 12036503a..4612fda23 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4888,7 +4888,7 @@ checksum = "6373607a59f0be73a39b6fe456b8192fcc3585f602af20751600e974dd455e77" [[package]] name = "lithos-llm" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/lithos-llm?rev=a1e3fd37b7153870411701327ac117606753fe90#a1e3fd37b7153870411701327ac117606753fe90" +source = "git+https://github.com/lithoscomputer/lithos-llm?rev=55add4596b861a0623d00c3a54aa5c147c8d504b#55add4596b861a0623d00c3a54aa5c147c8d504b" dependencies = [ "async-trait", "aws-config", @@ -5869,7 +5869,7 @@ dependencies = [ [[package]] name = "pebble-agent" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=c91810fe51aece80359b9cd8efea971af0c46925#c91810fe51aece80359b9cd8efea971af0c46925" dependencies = [ "async-trait", "futures-util", @@ -5886,7 +5886,7 @@ dependencies = [ [[package]] name = "pebble-cli-core" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=c91810fe51aece80359b9cd8efea971af0c46925#c91810fe51aece80359b9cd8efea971af0c46925" dependencies = [ "anyhow", "async-trait", @@ -5915,7 +5915,7 @@ dependencies = [ [[package]] name = "pebble-coding-agent" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=c91810fe51aece80359b9cd8efea971af0c46925#c91810fe51aece80359b9cd8efea971af0c46925" dependencies = [ "async-trait", "futures-util", diff --git a/Cargo.toml b/Cargo.toml index c932182a1..fd625c341 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -93,7 +93,7 @@ insta = "1" fabro-test = { path = "lib/foundation/fabro-test" } # Provider-neutral LLM catalog and client. Pinned to a revision until 0.x is # published to crates.io. -lithos-llm = { git = "https://github.com/lithoscomputer/lithos-llm", rev = "a1e3fd37b7153870411701327ac117606753fe90", default-features = false } +lithos-llm = { git = "https://github.com/lithoscomputer/lithos-llm", rev = "55add4596b861a0623d00c3a54aa5c147c8d504b", default-features = false } # Deterministic OpenAI twin used by twin-mode E2E tests; the same revision # lithos-llm verifies its codecs against. twin-openai = { git = "https://github.com/lithoscomputer/twins", rev = "ca45f0e50a6716d716aa2f638ca3cf767e88f613" } @@ -122,9 +122,9 @@ sandbox-driver-testing = { git = "https://github.com/lithoscomputer/sandbox-driv # sandbox, so the pebble and sandbox-driver pins move independently. Pebble # pins the same lithos-llm rev as fabro, and its lockfile policy is that # every shared crate resolves to the version lithos-llm locks. -pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a" } -pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a", features = ["mcp", "search-providers"] } -pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a" } +pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "c91810fe51aece80359b9cd8efea971af0c46925" } +pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "c91810fe51aece80359b9cd8efea971af0c46925", features = ["mcp", "search-providers"] } +pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "c91810fe51aece80359b9cd8efea971af0c46925" } sentry = { version = "0.35", default-features = false, features = ["backtrace", "contexts", "ureq", "rustls"] } fork = "0.2" exec = "0.3" diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml index 57d8689ad..29abc9088 100644 --- a/docs/public/api-reference/fabro-api.yaml +++ b/docs/public/api-reference/fabro-api.yaml @@ -35,8 +35,8 @@ tags: description: Workflow definitions and execution - name: Workflow Versions description: Immutable, content-addressed workflow packages - - name: Billing - description: Token counts and billed totals + - name: Usage + description: Token counts and costs - name: Insights description: SQL query editor and history - name: Models @@ -3714,21 +3714,21 @@ paths: schema: $ref: "#/components/schemas/ErrorResponse" - /api/v1/runs/{id}/billing: + /api/v1/runs/{id}/usage: get: - operationId: retrieveRunBilling + operationId: retrieveRunUsage tags: [Run Outputs] - summary: Retrieve Run Billing - description: Returns token counts and billed totals broken down by stage and model for a specific run. + summary: Retrieve Run Usage + description: Returns token counts and costs broken down by stage and model for a specific run. parameters: - $ref: "#/components/parameters/RunId" responses: "200": - description: Billing data + description: Usage data content: application/json: schema: - $ref: "#/components/schemas/RunBilling" + $ref: "#/components/schemas/RunUsage" "404": description: Run not found headers: @@ -5224,21 +5224,21 @@ paths: schema: $ref: "#/components/schemas/PaginatedHistoryEntryList" - # ── Billing ────────────────────────────────────────────────────────── + # ── Usage ──────────────────────────────────────────────────────────── - /api/v1/billing: + /api/v1/usage: get: - operationId: getAggregateBilling - tags: [Billing] - summary: Aggregate Billing - description: Returns aggregate token counts and billed totals across all completed runs since server start. + operationId: getAggregateUsage + tags: [Usage] + summary: Aggregate Usage + description: Returns aggregate token counts and costs across all completed runs since server start. responses: "200": - description: Aggregate billing data + description: Aggregate usage data content: application/json: schema: - $ref: "#/components/schemas/AggregateBilling" + $ref: "#/components/schemas/AggregateUsage" # ── System ─────────────────────────────────────────────────────────── @@ -8815,7 +8815,7 @@ components: $ref: "#/components/schemas/ReasoningEffort" description: Reasoning effort level. speed: - $ref: "#/components/schemas/BillingSpeed" + $ref: "#/components/schemas/Speed" description: Requested speed tier. metadata: type: object @@ -8827,38 +8827,45 @@ components: description: Raw provider options keyed by provider id. additionalProperties: true - CompletionUsage: + TokenCounts: description: > - lithos `TokenCounts`: five disjoint token buckets for one completion. - `input` excludes cache reads and writes, while `output` excludes - reasoning tokens when the provider reports them separately. + lithos `TokenCounts`: five disjoint token buckets. Every token is + counted in exactly one, so their plain sum is the total. `input` + excludes cache reads and writes, while `output` excludes reasoning + tokens when the provider reports them separately. A bucket that is + absent reads as zero. type: object properties: input: type: integer - format: int64 + format: uint64 + minimum: 0 default: 0 - description: Uncached prompt tokens. + description: Prompt tokens that were neither read from nor written to a cache. output: type: integer - format: int64 + format: uint64 + minimum: 0 default: 0 - description: Non-reasoning completion tokens. + description: Completion tokens that are not reasoning tokens. reasoning: type: integer - format: int64 + format: uint64 + minimum: 0 default: 0 - description: Separately reported reasoning tokens. + description: Completion tokens spent on reasoning, priced at the output rate. cache_read: type: integer - format: int64 + format: uint64 + minimum: 0 default: 0 description: Prompt tokens served from a provider cache. cache_write: type: integer - format: int64 + format: uint64 + minimum: 0 default: 0 - description: Prompt tokens written to a provider cache. + description: Prompt tokens written into a provider cache. ModelHandle: description: A resolved provider and model identity. @@ -8871,22 +8878,48 @@ components: type: string description: Canonical model id within the provider. - CompletionCost: + Cost: description: "lithos `Cost`: a USD amount in micros and where it came from." type: object required: [usd_micros, source] properties: usd_micros: type: integer - format: int64 + format: uint64 minimum: 0 source: $ref: "#/components/schemas/CostSource" + Usage: + description: >- + lithos `Usage`: token counts and, when known, what they cost. `cost` + is absent when there is no cost data, never zero. A sum has a cost + only when every part that used tokens was priced; its `source` is the + parts' shared source, or `application` when they differ. + type: object + required: [tokens] + properties: + tokens: + $ref: "#/components/schemas/TokenCounts" + cost: + $ref: "#/components/schemas/Cost" + + ModelUsage: + description: >- + Usage grouped under one model: one response, or one model's share of + a stage. + type: object + required: [model, usage] + properties: + model: + $ref: "#/components/schemas/UsageModelRef" + usage: + $ref: "#/components/schemas/Usage" + CompletionResponse: description: >- - A lithos `Response`, returned verbatim. The server is the billing - authority: `cost` is the catalog estimate or the provider's own + A lithos `Response`, returned verbatim. The server prices the + response: `cost` is the catalog estimate or the provider's own figure. When the request carried `schema`, `output` holds the parsed object. type: object @@ -8912,9 +8945,9 @@ components: type: string description: "Why generation stopped: stop, length, tool_call, content_filter, error, incomplete, or a provider-specific reason." usage: - $ref: "#/components/schemas/CompletionUsage" + $ref: "#/components/schemas/TokenCounts" cost: - $ref: "#/components/schemas/CompletionCost" + $ref: "#/components/schemas/Cost" rate_limits: type: object additionalProperties: true @@ -8935,7 +8968,8 @@ components: type: string description: > Where a cost came from: `catalog` (estimated from catalog prices), - `provider` (the provider's own billing data), or `application`. + `provider` (the provider's own reported cost), or `application` + (a sum the caller assembled from differently sourced parts). enum: [catalog, provider, application] PaginatedSavedQueryList: @@ -10424,7 +10458,7 @@ components: - type: "null" speed: oneOf: - - $ref: "#/components/schemas/BillingSpeed" + - $ref: "#/components/schemas/Speed" - type: "null" permission_level: oneOf: @@ -11159,10 +11193,15 @@ components: Open tool batch: when the batch started and which calls have not yet reported completion. usage: - $ref: "#/components/schemas/BilledTokenCounts" + $ref: "#/components/schemas/Usage" + description: >- + The stage's usage: while the stage runs, its agent's own + accounting of the session tree with whatever cost the provider + reported; once it ends, the same tokens with the catalog's price + where the provider reported none. model: oneOf: - - $ref: "#/components/schemas/BillingModelRef" + - $ref: "#/components/schemas/UsageModelRef" - type: "null" permission_level: oneOf: @@ -11191,18 +11230,18 @@ components: Start of an external ACP agent process, if one is running. ACP agents do not expose Fabro's internal LLM brackets, so the process lifetime supplies their live inference estimate. - billing_by_model: + usage_by_model: type: array items: - $ref: "#/components/schemas/BilledModelUsage" + $ref: "#/components/schemas/ModelUsage" default: [] description: >- The completed stage's `usage` split by model, as `stage.completed` reported it: the root session's route and each subagent's own - model, a subagent whose model the catalog does not know billed at + model, a subagent whose model the catalog does not know priced at the root's. Sums to `usage`. Empty while the stage runs and for - stages without a coding agent; the billing rollup then bills - `usage` to `model`. + stages without a coding agent; the usage rollup then puts `usage` + under `model`. agent: oneOf: - $ref: "#/components/schemas/AgentSessionProjection" @@ -11418,15 +11457,14 @@ components: provider-reported cost for the root session and each descendant, the route and where it moved, the context window, tools, MCP servers, skills, todo lists, subagents, compactions, files touched, and the - prompt in progress. Counts only; pricing a count from the catalog is - fabro's, and lives in `StageProjection.usage`. + prompt in progress. Its costs are the provider's own; pricing from + the catalog is fabro's, and lives in `StageProjection.usage`. type: object required: - root_session_id - route - activity - usage - - cost_usd_micros - messages - descendants - context_window @@ -11450,13 +11488,10 @@ components: activity: $ref: "#/components/schemas/AgentSessionActivity" usage: - $ref: "#/components/schemas/TokenUsage" - description: The root session's usage over the stage. - cost_usd_micros: - type: ["integer", "null"] - format: uint64 - minimum: 0 - description: The root session's provider-reported cost, when a provider reported one. + $ref: "#/components/schemas/Usage" + description: >- + The root session's usage over the stage, with the provider's + reported cost when every answer carried one. messages: type: integer format: uint64 @@ -11573,7 +11608,6 @@ components: required: - parent - usage - - cost_usd_micros - messages - compactions properties: @@ -11589,11 +11623,7 @@ components: The model it runs on, from its `SessionStarted`; when the start was not seen, the model of its first answer. usage: - $ref: "#/components/schemas/TokenUsage" - cost_usd_micros: - type: ["integer", "null"] - format: uint64 - minimum: 0 + $ref: "#/components/schemas/Usage" messages: type: integer format: uint64 @@ -11826,16 +11856,11 @@ components: type: integer minimum: 0 usage: - $ref: "#/components/schemas/TokenUsage" + $ref: "#/components/schemas/Usage" description: >- - The summary call's tokens: a breakdown of the session's and the - prompt's usage, which already include them. Zero on compactions + The summary call's usage: a breakdown of the session's and the + prompt's usage, which already include it. Zero on compactions recorded before it was kept. - cost_usd_micros: - type: integer - format: uint64 - minimum: 0 - description: The summary call's provider-reported cost, included in the totals the same way. AgentSessionRouteFailover: description: One move the root session made to a fallback route, as the stream reported it from the route it moved to. @@ -11865,15 +11890,11 @@ components: $ref: "#/components/schemas/AgentErrorData" description: The failure that ended the previous route. usage: - $ref: "#/components/schemas/TokenUsage" + $ref: "#/components/schemas/Usage" description: >- What the prompt spent on the failed route. Already in the session's and the prompt's totals through that route's committed answers: a breakdown, not an addition. - cost_usd_micros: - type: integer - format: uint64 - minimum: 0 inference_ms: type: integer format: uint64 @@ -11918,7 +11939,6 @@ components: required: - completed - usage - - cost_usd_micros - messages - context_window - tool_calls @@ -11932,12 +11952,8 @@ components: type: boolean description: Whether the prompt reached its end. usage: - $ref: "#/components/schemas/TokenUsage" + $ref: "#/components/schemas/Usage" description: The root session's usage over the prompt. - cost_usd_micros: - type: ["integer", "null"] - format: uint64 - minimum: 0 messages: type: integer format: uint64 @@ -11986,44 +12002,6 @@ components: last_file_touched: type: ["string", "null"] - TokenUsage: - description: >- - Token accounting as the coding agent counts it. The five buckets are - disjoint: every token is counted in exactly one, so their plain sum - is the total. A bucket that is absent reads as zero. - type: object - properties: - input: - type: integer - format: uint64 - minimum: 0 - default: 0 - description: Prompt tokens that were neither read from nor written to a cache. - output: - type: integer - format: uint64 - minimum: 0 - default: 0 - description: Completion tokens that are not reasoning tokens. - reasoning: - type: integer - format: uint64 - minimum: 0 - default: 0 - description: Completion tokens spent on reasoning, billed at the output rate. - cache_read: - type: integer - format: uint64 - minimum: 0 - default: 0 - description: Prompt tokens served from a provider cache. - cache_write: - type: integer - format: uint64 - minimum: 0 - default: 0 - description: Prompt tokens written into a provider cache. - McpToolSummary: description: One tool an MCP server advertised, as the coding agent's registry named it. type: object @@ -12207,7 +12185,7 @@ components: - type: "null" speed: oneOf: - - $ref: "#/components/schemas/BillingSpeed" + - $ref: "#/components/schemas/Speed" - type: "null" InterviewOption: @@ -12411,6 +12389,7 @@ components: - stage_id - stage_label - timing + - usage - retries properties: stage_id: @@ -12419,9 +12398,9 @@ components: type: string timing: $ref: "#/components/schemas/StageTiming" - billing_usd_micros: - type: ["integer", "null"] - format: int64 + usage: + $ref: "#/components/schemas/Usage" + description: Per-node usage summed across every visit of the node. retries: type: integer format: uint32 @@ -12455,10 +12434,13 @@ components: type: array items: $ref: "#/components/schemas/StageSummary" - billing: + usage: oneOf: - - $ref: "#/components/schemas/BilledTokenCounts" + - $ref: "#/components/schemas/Usage" - type: "null" + description: >- + The run's usage summed across every stage visit; null for a run + that made no model calls. total_retries: type: integer format: uint32 @@ -12654,7 +12636,7 @@ components: - source_directory - timestamps - timing - - billing + - usage - size - ask_fabro - diff @@ -12718,10 +12700,11 @@ components: description: | Run-level timing rollup. Wall time is the run's clock duration; active timing sums work across stage visits. - billing: - oneOf: - - $ref: "#/components/schemas/RunBillingSummary" - - type: "null" + usage: + $ref: "#/components/schemas/Usage" + description: >- + The run's usage summed across every stage visit so far: the + conclusion's total once the run ended, else the sum of the stages'. size: $ref: "#/components/schemas/RunSize" ask_fabro: @@ -12886,18 +12869,10 @@ components: type: ["string", "null"] format: date-time - RunBillingSummary: - type: object - required: [total_usd_micros] - properties: - total_usd_micros: - type: ["integer", "null"] - format: int64 - RunSize: type: string enum: [XS, S, M, L, XL] - description: Run size bucket derived from current best-effort billed usage. + description: Run size bucket derived from the run's current cost. RunLinks: type: object @@ -13089,75 +13064,10 @@ components: type: string enum: [github, git, unknown] - BilledTokenCounts: - description: Token counts with optional billed USD micros totals. - type: object - required: - - input_tokens - - output_tokens - - total_tokens - - reasoning_tokens - - cache_read_tokens - - cache_write_tokens - properties: - input_tokens: - type: integer - format: int64 - description: Number of input tokens consumed. - example: 28640 - output_tokens: - type: integer - format: int64 - description: Number of output tokens generated. - example: 8750 - total_tokens: - type: integer - format: int64 - description: Total billable tokens aggregated across categories. - example: 37390 - reasoning_tokens: - type: integer - format: int64 - description: Number of reasoning tokens. - example: 1200 - cache_read_tokens: - type: integer - format: int64 - description: Number of cache read tokens. - example: 4800 - cache_write_tokens: - type: integer - format: int64 - description: Number of cache write tokens. - example: 1500 - total_usd_micros: - type: ["integer", "null"] - format: int64 - description: Billed USD amount in micros. - example: 720000 - - BilledModelUsage: + UsageModelRef: description: >- - Usage and cost billed to one model: one response, or one model's share - of a stage. - type: object - required: - - model - - tokens - properties: - model: - $ref: "#/components/schemas/BillingModelRef" - tokens: - $ref: "#/components/schemas/CompletionUsage" - total_usd_micros: - type: integer - format: int64 - description: >- - Cost for `tokens`, when the provider reported one or the catalog - could price them. Absent means no cost data, not zero. - - BillingModelRef: - description: Provider-qualified billing model identity used for cost estimates. + Provider-qualified model identity a usage is grouped under. Carries + the requested speed tier because providers price tiers differently. type: object required: - provider @@ -13169,10 +13079,10 @@ components: type: string speed: oneOf: - - $ref: "#/components/schemas/BillingSpeed" + - $ref: "#/components/schemas/Speed" - type: "null" - BillingSpeed: + Speed: description: "lithos `Speed`: the requested latency or cost tier." type: string enum: @@ -13709,52 +13619,21 @@ components: description: Question text. example: Accept or push for another round? - AggregateBillingTotals: - description: Aggregate billing totals across all runs. + AggregateUsageTotals: + description: Aggregate usage totals across all runs. type: object required: - runs - - input_tokens - - output_tokens - - total_tokens - - reasoning_tokens - - cache_read_tokens - - cache_write_tokens + - usage - timing properties: runs: type: integer description: Total number of completed runs. example: 9 - input_tokens: - type: integer - description: Total input tokens. - example: 643860 - output_tokens: - type: integer - description: Total output tokens. - example: 189720 - total_tokens: - type: integer - description: Total tokens aggregated across all billing categories. - example: 833580 - reasoning_tokens: - type: integer - description: Total reasoning tokens. - example: 12040 - cache_read_tokens: - type: integer - description: Total cache read tokens. - example: 85400 - cache_write_tokens: - type: integer - description: Total cache write tokens. - example: 9200 - total_usd_micros: - type: ["integer", "null"] - format: int64 - description: Total billed USD amount in micros. - example: 20340000 + usage: + $ref: "#/components/schemas/Usage" + description: Tokens and cost summed across every completed run. timing: $ref: "#/components/schemas/RunTiming" description: | @@ -13762,8 +13641,8 @@ components: sums work across stage visits, so `active_time_ms` can exceed `wall_time_ms`. - BillingStageRef: - description: Reference to a workflow node in a billing stage row. + UsageStageRef: + description: Reference to a workflow node in a usage stage row. type: object required: - id @@ -13881,7 +13760,7 @@ components: - status - node_id - visit - - billing + - usage properties: id: $ref: "#/components/schemas/StageId" @@ -13957,15 +13836,15 @@ components: format: date-time description: Wall-clock time the latest attempt of this stage started, if known. example: "2026-04-29T12:34:56Z" - billing: - $ref: "#/components/schemas/BilledTokenCounts" + usage: + $ref: "#/components/schemas/Usage" description: >- - Token counts for this stage execution alone. `total_usd_micros` is - the provider-reported cost when there is one, otherwise the server - catalog's price for these tokens — the same pricing the - `/runs/{id}/billing` rows use. All-zero counts mean the stage made - no model calls. Unlike the billing rows, which sum every visit of a - node, this covers only this visit. + Usage for this stage execution alone. `cost` is the provider's + reported cost when there is one, otherwise the server catalog's + price for these tokens — the same pricing the `/runs/{id}/usage` + rows use. All-zero counts mean the stage made no model calls. + Unlike the usage rows, which sum every visit of a node, this + covers only this visit. # ── File Diff Schemas ────────────────────────────────────────────── @@ -14300,26 +14179,26 @@ components: meta: $ref: "#/components/schemas/RunCommitsMeta" - # ── Billing Schemas ────────────────────────────────────────────────── + # ── Usage Schemas ──────────────────────────────────────────────────── - RunBillingStage: - description: Token counts and billed totals for one workflow node within a run. Rows are grouped by node; billing and timing sum every visit of that node. + RunUsageStage: + description: Token counts and cost for one workflow node within a run. Rows are grouped by node; usage and timing sum every visit of that node. type: object required: - stage - model - - billing + - usage - timing properties: stage: - $ref: "#/components/schemas/BillingStageRef" + $ref: "#/components/schemas/UsageStageRef" model: description: Latest usage-bearing visit model for this node; null when no visit used an LLM model. oneOf: - - $ref: "#/components/schemas/BillingModelRef" + - $ref: "#/components/schemas/UsageModelRef" - type: "null" - billing: - $ref: "#/components/schemas/BilledTokenCounts" + usage: + $ref: "#/components/schemas/Usage" timing: $ref: "#/components/schemas/StageTiming" description: | @@ -14336,72 +14215,43 @@ components: - type: "null" description: Lifecycle state of the stage. Use to detect in-flight rows for client-side runtime ticking. - RunBillingTotals: - description: Aggregate billing totals across all stages of a run. + RunUsageTotals: + description: Aggregate usage totals across all stages of a run. type: object required: - timing - - input_tokens - - output_tokens - - total_tokens - - reasoning_tokens - - cache_read_tokens - - cache_write_tokens + - usage properties: timing: $ref: "#/components/schemas/RunTiming" description: | Run-level timing rollup. `wall_time_ms` is summed across stage visits; active timing sums work across visits. - input_tokens: - type: integer - description: Total input tokens consumed. - example: 71540 - output_tokens: - type: integer - description: Total output tokens generated. - example: 21080 - total_tokens: - type: integer - description: Total tokens aggregated across all billing categories. - example: 92620 - reasoning_tokens: - type: integer - description: Total reasoning tokens. - example: 3400 - cache_read_tokens: - type: integer - description: Total cache read tokens. - example: 22000 - cache_write_tokens: - type: integer - description: Total cache write tokens. - example: 4500 - total_usd_micros: - type: ["integer", "null"] - format: int64 - description: Total billed USD amount in micros. - example: 2260000 + usage: + $ref: "#/components/schemas/Usage" + description: >- + Tokens and cost summed across every stage visit. The cost is + known only when every visit that used tokens was priced. - BillingByModel: - description: Billing statistics grouped by model. + UsageByModel: + description: Usage grouped by model. type: object required: - model - stages - - billing + - usage properties: model: - $ref: "#/components/schemas/BillingModelRef" + $ref: "#/components/schemas/UsageModelRef" stages: type: integer description: Number of usage-bearing stage visits that used this model. example: 2 - billing: - $ref: "#/components/schemas/BilledTokenCounts" + usage: + $ref: "#/components/schemas/Usage" - RunBilling: - description: Complete billing breakdown for a single run. + RunUsage: + description: Complete usage breakdown for a single run. type: object required: - stages @@ -14410,31 +14260,31 @@ components: properties: stages: type: array - description: Per-node billing breakdown. Each row sums billing and runtime across all visits of that node. + description: Per-node usage breakdown. Each row sums usage and runtime across all visits of that node. items: - $ref: "#/components/schemas/RunBillingStage" + $ref: "#/components/schemas/RunUsageStage" totals: - $ref: "#/components/schemas/RunBillingTotals" + $ref: "#/components/schemas/RunUsageTotals" by_model: type: array - description: Billing grouped by model. + description: Usage grouped by model. items: - $ref: "#/components/schemas/BillingByModel" + $ref: "#/components/schemas/UsageByModel" - AggregateBilling: - description: Aggregate token counts and billed totals across all runs since server start. + AggregateUsage: + description: Aggregate token counts and costs across all runs since server start. type: object required: - totals - by_model properties: totals: - $ref: "#/components/schemas/AggregateBillingTotals" + $ref: "#/components/schemas/AggregateUsageTotals" by_model: type: array - description: Billing grouped by model. + description: Usage grouped by model. items: - $ref: "#/components/schemas/BillingByModel" + $ref: "#/components/schemas/UsageByModel" PreviewUrlRequest: description: Request body for generating a preview URL from a sandbox port. diff --git a/lib/apps/fabro-cli/src/commands/run/events.rs b/lib/apps/fabro-cli/src/commands/run/events.rs index ffebde99a..062c7fc10 100644 --- a/lib/apps/fabro-cli/src/commands/run/events.rs +++ b/lib/apps/fabro-cli/src/commands/run/events.rs @@ -390,8 +390,11 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O "succeeded" | "partially_succeeded" => &styles.bold_green, _ => &styles.bold_red, }; + let usage = prop_field(envelope, "usage"); let cost = format_cost( - prop_field(envelope, "total_usd_micros") + usage + .and_then(|value| value.get("cost")) + .and_then(|value| value.get("usd_micros")) .or_else(|| prop_field(envelope, "total_cost")), ); @@ -407,13 +410,12 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O let mut lines = vec![summary]; - if let Some(billing) = - prop_field(envelope, "billing").or_else(|| prop_field(envelope, "usage")) - { - let total = billing - .get("total_tokens") - .and_then(serde_json::Value::as_u64) - .unwrap_or(0); + if let Some(tokens) = usage.and_then(|value| value.get("tokens")) { + let bucket = |name: &str| tokens.get(name).and_then(serde_json::Value::as_u64); + let total = ["input", "output", "reasoning", "cache_read", "cache_write"] + .into_iter() + .filter_map(bucket) + .fold(0_u64, u64::saturating_add); let pad = " ".repeat(ts.len() + 1); if total > 0 { lines.push(format!( @@ -424,14 +426,8 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O .apply_to(format!("Tokens: {}", format_tokens(total))) )); } - if let Some(cache_read) = billing - .get("cache_read_tokens") - .and_then(serde_json::Value::as_u64) - { - let cache_write = billing - .get("cache_write_tokens") - .and_then(serde_json::Value::as_u64) - .unwrap_or(0); + if let Some(cache_read) = bucket("cache_read") { + let cache_write = bucket("cache_write").unwrap_or(0); lines.push(format!( "{}{}", pad, @@ -442,10 +438,7 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O )) )); } - if let Some(reasoning) = billing - .get("reasoning_tokens") - .and_then(serde_json::Value::as_u64) - { + if let Some(reasoning) = bucket("reasoning") { if reasoning > 0 { lines.push(format!( "{}{}", @@ -535,18 +528,20 @@ fn format_event_pretty_value(envelope: &serde_json::Value, styles: &Styles) -> O "stage.completed" => { let label = str_field(envelope, "node_label").unwrap_or("?"); let duration = format_duration_ms(timing_wall_field(envelope)); - let billing = prop_field(envelope, "billing").or_else(|| prop_field(envelope, "usage")); + // `stage.completed.usage` is a `ModelUsage`: the model, then the usage. + let usage = prop_field(envelope, "usage").and_then(|value| value.get("usage")); let cost = format_cost( - billing - .and_then(|value| value.get("total_usd_micros")) - .or_else(|| billing.and_then(|value| value.get("cost"))), + usage + .and_then(|value| value.get("cost")) + .and_then(|value| value.get("usd_micros")), ); - let input_tokens = billing - .and_then(|value| value.get("input_tokens")) + let tokens = usage.and_then(|value| value.get("tokens")); + let input_tokens = tokens + .and_then(|value| value.get("input")) .and_then(serde_json::Value::as_u64) .unwrap_or(0); - let output_tokens = billing - .and_then(|value| value.get("output_tokens")) + let output_tokens = tokens + .and_then(|value| value.get("output")) .and_then(serde_json::Value::as_u64) .unwrap_or(0); let token_total = input_tokens.saturating_add(output_tokens); @@ -921,7 +916,7 @@ fn format_duration_ms(value: Option<&serde_json::Value>) -> String { fn format_cost(value: Option<&serde_json::Value>) -> String { match value { Some(value) => { - if let Some(usd_micros) = value.as_i64() { + if let Some(usd_micros) = value.as_u64() { if usd_micros > 0 { return format_usd_micros(usd_micros); } @@ -1091,7 +1086,7 @@ mod tests { #[test] fn pretty_stage_completed() { let styles = no_color_styles(); - let line = r#"{"ts":"2026-01-01T14:23:15Z","event":"stage.completed","node_label":"plan","properties":{"timing":{"wall_time_ms":8000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"cost":0.12,"input_tokens":10000,"output_tokens":5200}}}"#; + let line = r#"{"ts":"2026-01-01T14:23:15Z","event":"stage.completed","node_label":"plan","properties":{"timing":{"wall_time_ms":8000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"model":{"provider":"openai","model_id":"gpt-5.4"},"usage":{"tokens":{"input":10000,"output":5200},"cost":{"usd_micros":120000,"source":"catalog"}}}}}"#; let result = format_event_pretty(line, &styles).unwrap(); assert!(result.contains("plan"), "got: {result}"); assert!(result.contains("$0.12"), "got: {result}"); @@ -1170,12 +1165,12 @@ mod tests { #[test] fn pretty_workflow_run_completed() { let styles = no_color_styles(); - let line = r#"{"ts":"2026-01-01T14:23:32Z","run_id":"abc123","event":"run.completed","properties":{"timing":{"wall_time_ms":25000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","total_usd_micros":570000,"billing":{"input_tokens":5000,"output_tokens":2000,"total_tokens":7000,"cache_read_tokens":3000,"cache_write_tokens":500,"reasoning_tokens":800}}}"#; + let line = r#"{"ts":"2026-01-01T14:23:32Z","run_id":"abc123","event":"run.completed","properties":{"timing":{"wall_time_ms":25000,"inference_time_ms":0,"tool_time_ms":0,"active_time_ms":0},"status":"succeeded","usage":{"tokens":{"input":5000,"output":2000,"cache_read":3000,"cache_write":500,"reasoning":800},"cost":{"usd_micros":570000,"source":"catalog"}}}}"#; let result = format_event_pretty(line, &styles).unwrap(); assert!(result.contains("SUCCEEDED"), "got: {result}"); assert!(result.contains("25s"), "got: {result}"); assert!(result.contains("$0.57"), "got: {result}"); - assert!(result.contains("7.0k toks"), "got: {result}"); + assert!(result.contains("11.3k toks"), "got: {result}"); assert!(result.contains("Cache:"), "got: {result}"); assert!(result.contains("3.0k toks read"), "got: {result}"); assert!(result.contains("Reasoning:"), "got: {result}"); diff --git a/lib/apps/fabro-cli/src/commands/run/output.rs b/lib/apps/fabro-cli/src/commands/run/output.rs index 9de4a3158..c05d7c537 100644 --- a/lib/apps/fabro-cli/src/commands/run/output.rs +++ b/lib/apps/fabro-cli/src/commands/run/output.rs @@ -208,10 +208,11 @@ pub(crate) fn print_run_conclusion( HumanDuration(Duration::from_millis(conclusion.timing.wall_time_ms)) ); - if let Some(billing) = conclusion.billing.as_ref() { - let total_tokens = billing.total_tokens; + if let Some(usage) = conclusion.usage { + let total_tokens = usage.total_tokens(); + let cost_usd_micros = usage.cost.map(|cost| cost.usd_micros); if total_tokens > 0 { - if let Some(total_usd_micros) = billing.total_usd_micros { + if let Some(total_usd_micros) = cost_usd_micros { if total_usd_micros > 0 { fabro_util::printerr!( printer, @@ -232,28 +233,28 @@ pub(crate) fn print_run_conclusion( .apply_to(format!("Toks: {}", format_tokens_human(total_tokens))) ); } - if billing.cache_read_tokens > 0 || billing.cache_write_tokens > 0 { + if usage.tokens.cache_read > 0 || usage.tokens.cache_write > 0 { fabro_util::printerr!( printer, "{}", styles.dim.apply_to(format!( "Cache: {} read, {} write", - format_tokens_human(billing.cache_read_tokens), - format_tokens_human(billing.cache_write_tokens), + format_tokens_human(usage.tokens.cache_read), + format_tokens_human(usage.tokens.cache_write), )), ); } - if billing.reasoning_tokens > 0 { + if usage.tokens.reasoning > 0 { fabro_util::printerr!( printer, "{}", styles.dim.apply_to(format!( "Reasoning: {} tokens", - format_tokens_human(billing.reasoning_tokens), + format_tokens_human(usage.tokens.reasoning), )), ); } - } else if billing.total_usd_micros.is_none() { + } else if cost_usd_micros.is_none() { fabro_util::printerr!( printer, "{}", diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs index 7d9377dde..e98535eac 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/event.rs @@ -1,5 +1,5 @@ use chrono::{DateTime, Utc}; -use fabro_types::{BilledModelUsage, EventBody, RunEvent}; +use fabro_types::{EventBody, ModelUsage, RunEvent}; use fabro_util::{error, text}; use fabro_workflow::event::RunNoticeLevel; use pebble_coding_agent::events::{CodingEvent, ErrorKind as AgentErrorKind, LlmOutputKind}; @@ -13,12 +13,15 @@ pub(super) struct ProgressUsage { } impl ProgressUsage { - pub(super) fn from_stage_usage(usage: &BilledModelUsage) -> Self { - let tokens = usage.tokens(); + pub(super) fn from_stage_usage(usage: &ModelUsage) -> Self { + let tokens = usage.usage.tokens; Self { input_tokens: tokens.input, output_tokens: tokens.billable_output(), - cost: usage.total_usd_micros.map(|cost| cost as f64 / 1_000_000.0), + cost: usage + .usage + .cost + .map(|cost| cost.usd_micros as f64 / 1_000_000.0), } } @@ -294,7 +297,7 @@ pub(super) fn from_run_event(stored: &RunEvent) -> Option { name: node_label, timing: props.timing, status: props.status.to_string(), - usage: props.billing.as_ref().map(ProgressUsage::from_stage_usage), + usage: props.usage.as_ref().map(ProgressUsage::from_stage_usage), }), EventBody::StageFailed(props) => Some(ProgressEvent::StageFailed { node_id, @@ -651,8 +654,8 @@ mod tests { status: "succeeded".into(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs index ca3f5a020..b5dd8876e 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/mod.rs @@ -461,12 +461,12 @@ mod tests { use fabro_workflow::event::{ Event, RunNoticeLevel, SandboxLifecycle, to_run_event, to_run_event_at, }; - use fabro_workflow::outcome::billed_model_usage_from_llm; + use fabro_workflow::outcome::model_usage_from_llm; use lithos_llm::catalog::{ModelId, builtin}; use lithos_llm::types::TokenCounts; use pebble_coding_agent::events::{ CodingAgentEvent, CodingEvent, CompactionReason, ErrorData as AgentErrorData, - ErrorKind as AgentErrorKind, TokenUsage, + ErrorKind as AgentErrorKind, Usage, }; use super::*; @@ -618,9 +618,7 @@ mod tests { CodingEvent::AssistantMessage { text: text.into(), model: model.into(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window: None, reasoning: None, @@ -650,9 +648,9 @@ mod tests { status: "succeeded".into(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: Some( - billed_model_usage_from_llm( + usage_by_model: Vec::new(), + usage: Some( + model_usage_from_llm( &fabro_llm::test_support::test_catalog(), &ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")), TokenCounts { @@ -781,8 +779,7 @@ mod tests { summary_token_estimate: 500, tracked_file_count: 3, reason: CompactionReason::Threshold, - usage: TokenUsage::default(), - cost_usd_micros: None, + usage: Usage::default(), }), ); assert!(ui.stage.active_stages["s1"].compaction_bar.is_none()); diff --git a/lib/apps/fabro-cli/src/commands/run/run_progress/stage_display.rs b/lib/apps/fabro-cli/src/commands/run/run_progress/stage_display.rs index a6aa5ba60..cf34bb473 100644 --- a/lib/apps/fabro-cli/src/commands/run/run_progress/stage_display.rs +++ b/lib/apps/fabro-cli/src/commands/run/run_progress/stage_display.rs @@ -161,7 +161,6 @@ impl StageDisplay { self.stage_counts.get(node_id).copied().unwrap_or((0, 0)); let total_tokens = usage.map_or(0, ProgressUsage::total_tokens); if turn_count > 0 || tool_call_count > 0 || total_tokens > 0 { - let total_tokens = i64::try_from(total_tokens).unwrap_or(i64::MAX); format!( " {}", renderer.styles().dim.apply_to(format!( diff --git a/lib/apps/fabro-cli/src/commands/run/runner.rs b/lib/apps/fabro-cli/src/commands/run/runner.rs index e96388091..b67fd873d 100644 --- a/lib/apps/fabro-cli/src/commands/run/runner.rs +++ b/lib/apps/fabro-cli/src/commands/run/runner.rs @@ -1341,11 +1341,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, })), Some(WorkerTitlePhase::Succeeded) ); @@ -1359,7 +1358,7 @@ mod tests { final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, })), Some(WorkerTitlePhase::Cancelled) ); @@ -1373,7 +1372,7 @@ mod tests { final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, })), Some(WorkerTitlePhase::Failed) ); diff --git a/lib/apps/fabro-cli/src/commands/run/wait.rs b/lib/apps/fabro-cli/src/commands/run/wait.rs index 3e2017c36..feccf3678 100644 --- a/lib/apps/fabro-cli/src/commands/run/wait.rs +++ b/lib/apps/fabro-cli/src/commands/run/wait.rs @@ -84,12 +84,9 @@ fn build_json_output( if let Some(c) = conclusion { value["timing"] = serde_json::to_value(c.timing).unwrap_or_else(|_| serde_json::Value::Null); - if let Some(total_usd_micros) = c - .billing - .as_ref() - .and_then(|billing| billing.total_usd_micros) - { - value["total_usd_micros"] = total_usd_micros.into(); + if let Some(usage) = c.usage { + value["usage"] = + serde_json::to_value(usage).unwrap_or_else(|_| serde_json::Value::Null); } } value @@ -117,10 +114,9 @@ fn print_human_output( Some(c) => { let duration = format_duration_ms(c.timing.wall_time_ms); let cost = c - .billing - .as_ref() - .and_then(|billing| billing.total_usd_micros) - .map(|value| format!(" {}", format_usd_micros(value))) + .usage + .and_then(|usage| usage.cost) + .map(|cost| format!(" {}", format_usd_micros(cost.usd_micros))) .unwrap_or_default(); format!(" {duration}{cost}") } @@ -138,13 +134,25 @@ fn print_human_output( #[cfg(test)] mod tests { use fabro_types::{ - BilledTokenCounts, FailureCategory, FailureDetail, FailureReason, RunDiff, RunFailure, - RunStatus, StageOutcome, SuccessReason, fixtures, + FailureCategory, FailureDetail, FailureReason, RunDiff, RunFailure, RunStatus, + StageOutcome, SuccessReason, fixtures, }; use fabro_workflow::records::Conclusion; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use super::*; + /// A usage with only a catalog cost. + fn priced(usd_micros: u64) -> Usage { + Usage { + tokens: TokenCounts::default(), + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), + } + } + fn no_color_styles() -> Styles { Styles::new(false) } @@ -159,15 +167,7 @@ mod tests { failure: None, final_git_commit_sha: None, stages: vec![], - billing: Some(BilledTokenCounts { - input_tokens: 0, - output_tokens: 0, - total_tokens: 0, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(420_000), - }), + usage: Some(priced(420_000)), total_retries: 0, diff: RunDiff::default(), }; @@ -181,7 +181,8 @@ mod tests { assert_eq!(json["run_id"], run_id.to_string()); assert_eq!(json["status"], "succeeded"); assert_eq!(json["timing"]["wall_time_ms"], 12345); - assert_eq!(json["total_usd_micros"], 420_000); + assert_eq!(json["usage"]["cost"]["usd_micros"], 420_000); + assert_eq!(json["usage"]["cost"]["source"], "catalog"); } #[test] @@ -197,7 +198,7 @@ mod tests { assert_eq!(json["run_id"], run_id.to_string()); assert_eq!(json["status"], "failed"); assert!(json.get("timing").is_none()); - assert!(json.get("total_usd_micros").is_none()); + assert!(json.get("usage").is_none()); } #[test] @@ -221,7 +222,7 @@ mod tests { }), final_git_commit_sha: None, stages: vec![], - billing: None, + usage: None, total_retries: 0, diff: RunDiff::default(), }; @@ -232,7 +233,7 @@ mod tests { &run_id, Some(&conclusion), ); - assert!(json.get("total_usd_micros").is_none()); + assert!(json.get("usage").is_none()); assert_eq!(json["timing"]["wall_time_ms"], 500); } @@ -247,15 +248,7 @@ mod tests { failure: None, final_git_commit_sha: None, stages: vec![], - billing: Some(BilledTokenCounts { - input_tokens: 0, - output_tokens: 0, - total_tokens: 0, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(150_000), - }), + usage: Some(priced(150_000)), total_retries: 0, diff: RunDiff::default(), }; diff --git a/lib/apps/fabro-cli/src/main.rs b/lib/apps/fabro-cli/src/main.rs index 3ea411a3d..ea7fb44db 100644 --- a/lib/apps/fabro-cli/src/main.rs +++ b/lib/apps/fabro-cli/src/main.rs @@ -362,7 +362,7 @@ async fn main_inner(worker_token: Option) -> (String, Result<()>) { Box::pin(commands::pr::dispatch(ns, &base_ctx)).await?; } Commands::Parent(ns) => { - commands::parent::dispatch(ns, &base_ctx).await?; + Box::pin(commands::parent::dispatch(ns, &base_ctx)).await?; } Commands::Secret(ns) => { commands::secret::dispatch(ns, &base_ctx).await?; diff --git a/lib/apps/fabro-cli/src/server_runs.rs b/lib/apps/fabro-cli/src/server_runs.rs index a2fe5c88f..239f7e001 100644 --- a/lib/apps/fabro-cli/src/server_runs.rs +++ b/lib/apps/fabro-cli/src/server_runs.rs @@ -77,11 +77,8 @@ impl ServerRunInfo { self.run.timing.as_ref().map(|t| t.wall_time_ms) } - pub(crate) fn total_usd_micros(&self) -> Option { - self.run - .billing - .as_ref() - .and_then(|billing| billing.total_usd_micros) + pub(crate) fn total_usd_micros(&self) -> Option { + self.run.usage.cost.map(|cost| cost.usd_micros) } pub(crate) fn source_directory(&self) -> Option<&str> { diff --git a/lib/apps/fabro-cli/src/shared/utilities.rs b/lib/apps/fabro-cli/src/shared/utilities.rs index c6b048a1d..606483fba 100644 --- a/lib/apps/fabro-cli/src/shared/utilities.rs +++ b/lib/apps/fabro-cli/src/shared/utilities.rs @@ -130,7 +130,7 @@ pub(crate) fn relative_path(path: &Path) -> String { tilde_path(path) } -pub(crate) fn format_tokens_human(tokens: i64) -> String { +pub(crate) fn format_tokens_human(tokens: u64) -> String { if tokens >= 1_000_000 { format!("{:.1}m", tokens as f64 / 1_000_000.0) } else if tokens >= 1000 { @@ -140,7 +140,7 @@ pub(crate) fn format_tokens_human(tokens: i64) -> String { } } -pub(crate) fn format_usd_micros(usd_micros: i64) -> String { +pub(crate) fn format_usd_micros(usd_micros: u64) -> String { format!("${:.2}", usd_micros as f64 / 1_000_000.0) } diff --git a/lib/apps/fabro-cli/tests/it/cmd/run.rs b/lib/apps/fabro-cli/tests/it/cmd/run.rs index b5a00d326..7e7188cb2 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/run.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/run.rs @@ -63,7 +63,7 @@ fn remote_run_state_response(run_id: &str) -> serde_json::Value { "status": "succeeded", "timing": {"wall_time_ms": 12, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, "stages": [], - "billing": null, + "usage": null, "total_retries": 0, "diff": {} }); diff --git a/lib/apps/fabro-cli/tests/it/cmd/support.rs b/lib/apps/fabro-cli/tests/it/cmd/support.rs index 5527027b7..1aeab1aa3 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/support.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/support.rs @@ -338,7 +338,7 @@ pub(crate) fn remote_run_summary_json( "completed_at": null }, "timing": null, - "billing": null, + "usage": {"tokens": {"input": 0, "output": 0, "reasoning": 0, "cache_read": 0, "cache_write": 0}}, "diff": null, "pull_request": null, "current_question": null, @@ -1253,10 +1253,9 @@ async fn append_seeded_simple_completion_events( "artifact_count": 0, "status": "succeeded", "reason": "completed", - "total_usd_micros": null, "final_git_commit_sha": null, "final_patch": null, - "billing": null, + "usage": null, }), ) .await; @@ -1428,10 +1427,9 @@ async fn append_seeded_git_completion_events( "artifact_count": 0, "status": "succeeded", "reason": "completed", - "total_usd_micros": null, "final_git_commit_sha": step_two_sha, "final_patch": final_story_patch(), - "billing": null, + "usage": null, }), ) .await; @@ -1498,10 +1496,9 @@ async fn append_seeded_git_noop_events( "artifact_count": 0, "status": "succeeded", "reason": "completed", - "total_usd_micros": null, "final_git_commit_sha": base_sha, "final_patch": null, - "billing": null, + "usage": null, }), ) .await; @@ -1567,10 +1564,9 @@ async fn append_seeded_artifact_run_events( "artifact_count": 7, "status": "succeeded", "reason": "completed", - "total_usd_micros": null, "final_git_commit_sha": null, "final_patch": null, - "billing": null, + "usage": null, }), ) .await; @@ -1728,7 +1724,7 @@ fn stage_completed_properties(index: usize, response: Option<&str>) -> serde_jso "status": "succeeded", "preferred_label": null, "suggested_next_ids": [], - "billing": null, + "usage": null, "failure": null, "notes": null, "files_touched": [], diff --git a/lib/apps/fabro-server/src/demo/mod.rs b/lib/apps/fabro-server/src/demo/mod.rs index 14936a897..c40fad278 100644 --- a/lib/apps/fabro-server/src/demo/mod.rs +++ b/lib/apps/fabro-server/src/demo/mod.rs @@ -335,12 +335,12 @@ fn demo_run_files() -> PaginatedRunFileList { } } -pub(crate) async fn get_run_billing( +pub(crate) async fn get_run_usage( _auth: RequiredUser, State(_state): State>, Path(_id): Path, ) -> Response { - (StatusCode::OK, Json(runs::billing())).into_response() + (StatusCode::OK, Json(runs::usage())).into_response() } pub(crate) async fn get_run_settings( @@ -1073,11 +1073,11 @@ pub(crate) async fn prune_runs( // ── Usage ────────────────────────────────────────────────────────────── -pub(crate) async fn get_aggregate_billing( +pub(crate) async fn get_aggregate_usage( _auth: RequiredUser, State(_state): State>, ) -> Response { - (StatusCode::OK, Json(billing::aggregate())).into_response() + (StatusCode::OK, Json(usage::aggregate())).into_response() } // ── Data modules ─────────────────────────────────────────────────────── @@ -1101,11 +1101,11 @@ mod runs { }; use fabro_types::settings::{InterpString, ProjectNamespace, WorkflowNamespace}; use fabro_types::{ - AuthMethod, IdpIdentity, PendingReason, Principal, RepositoryRef, RunBillingSummary, RunId, - RunLifecycle, RunLinks, RunOrigin, RunSize, RunTimestamps, StageId, WorkflowRef, - WorkflowSettings, + AuthMethod, IdpIdentity, PendingReason, Principal, RepositoryRef, RunId, RunLifecycle, + RunLinks, RunOrigin, RunSize, RunTimestamps, StageId, WorkflowRef, WorkflowSettings, }; use lithos_llm::catalog::ProviderId; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use super::ts; @@ -1124,14 +1124,29 @@ mod runs { .collect() } - fn billing_model(provider: ProviderId, model_id: &str) -> BillingModelRef { - BillingModelRef { + fn usage_model(provider: ProviderId, model_id: &str) -> UsageModelRef { + UsageModelRef { provider, model_id: model_id.into(), speed: None, } } + /// Demo usage priced from the catalog. + fn priced(input: u64, output: u64, usd_micros: u64) -> Usage { + Usage { + tokens: TokenCounts { + input, + output, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), + } + } + fn stage( stage_id: &StageId, name: &str, @@ -1143,7 +1158,7 @@ mod runs { id: stage_id.clone(), name: name.to_owned(), handler, - billing: BilledTokenCounts::default(), + usage: Usage::default(), status, wall_time_ms, node_id: stage_id.node_id().to_owned(), @@ -1188,7 +1203,7 @@ mod runs { elapsed_secs: Option, status_reason: Option<&str>, pending_control: Option, - total_usd_micros: Option, + cost_usd_micros: Option, entries: &[(&str, &str)], ) -> Run { let created_at = ts(created_at); @@ -1197,6 +1212,10 @@ mod runs { let repo_origin_url = Some(format!("https://github.com/demo/{repo_name}.git")); let wall_time_ms = elapsed_secs.and_then(duration_ms_from_secs); let timing = wall_time_ms.map(fabro_types::RunTiming::wall_only); + let cost = cost_usd_micros.map(|usd_micros| Cost { + usd_micros, + source: CostSource::Catalog, + }); Run { id: run_id, parent_id: None, @@ -1238,10 +1257,11 @@ mod runs { completed_at: Some(created_at), }, timing, - billing: total_usd_micros.map(|total_usd_micros| RunBillingSummary { - total_usd_micros: Some(total_usd_micros), - }), - size: RunSize::from_total_usd_micros(total_usd_micros), + usage: Usage { + tokens: TokenCounts::default(), + cost, + }, + size: RunSize::from_cost(cost), ask_fabro: Default::default(), diff: None, pull_request: None, @@ -1463,7 +1483,6 @@ mod runs { use pebble_coding_agent::events::{ CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind, FailoverContinuation, InputSource, McpToolSummary, SkillActivationSource, SkillSummary, - TokenUsage, }; let run_id = demo_run_id(1); @@ -1508,13 +1527,11 @@ mod runs { |model: &str, text: &str, input: u64, output: u64| CodingEvent::AssistantMessage { text: text.into(), model: model.into(), - usage: TokenUsage { + usage: Usage::from(TokenCounts { input, output, - ..TokenUsage::default() - }, - cost_usd_micros: None, - cost_source: None, + ..TokenCounts::default() + }), tool_call_count: 0, context_window: None, reasoning: None, @@ -1650,19 +1667,18 @@ mod runs { turns_used: 2, }), agent(CodingEvent::RouteFailover { - from: "anthropic/claude-opus-4.6".into(), - to: "openai/gpt-5.4".into(), - attempt: 1, - error: ErrorData::new(ErrorKind::Llm, "rate limited: retry after 30s"), - usage: TokenUsage { + from: "anthropic/claude-opus-4.6".into(), + to: "openai/gpt-5.4".into(), + attempt: 1, + error: ErrorData::new(ErrorKind::Llm, "rate limited: retry after 30s"), + usage: Usage::from(TokenCounts { input: 3_600, output: 540, - ..TokenUsage::default() - }, - cost_usd_micros: None, - inference_ms: 4_200, - tool_ms: 900, - continuation: FailoverContinuation::ContinueTurn, + ..TokenCounts::default() + }), + inference_ms: 4_200, + tool_ms: 900, + continuation: FailoverContinuation::ContinueTurn, }), agent(CodingEvent::CompactionCompleted { original_turn_count: 20, @@ -1670,12 +1686,11 @@ mod runs { summary_token_estimate: 500, tracked_file_count: 2, reason: CompactionReason::Threshold, - usage: TokenUsage { + usage: Usage::from(TokenCounts { input: 2_000, output: 500, - ..TokenUsage::default() - }, - cost_usd_micros: None, + ..TokenCounts::default() + }), }), call_started( "write_file", @@ -1757,153 +1772,91 @@ mod runs { projection } - pub(super) fn billing() -> RunBilling { - RunBilling { + pub(super) fn usage() -> RunUsage { + RunUsage { stages: vec![ - RunBillingStage { - stage: BillingStageRef { + RunUsageStage { + stage: UsageStageRef { id: "detect-drift".into(), name: "Detect Drift".into(), }, - model: Some(billing_model( + model: Some(usage_model( lithos_llm::catalog::builtin::anthropic(), "claude-opus-4-6", )), - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 12480, - output_tokens: 3210, - reasoning_tokens: 0, - total_tokens: 15690, - total_usd_micros: Some(480_000), - }, + usage: priced(12480, 3210, 480_000), timing: fabro_types::StageTiming::wall_only(72_000), started_at: None, state: Some(StageState::Succeeded), }, - RunBillingStage { - stage: BillingStageRef { + RunUsageStage { + stage: UsageStageRef { id: "propose-changes".into(), name: "Propose Changes".into(), }, - model: Some(billing_model( + model: Some(usage_model( lithos_llm::catalog::builtin::gemini(), "gemini-3.1-pro-preview", )), - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 28640, - output_tokens: 8750, - reasoning_tokens: 0, - total_tokens: 37390, - total_usd_micros: Some(720_000), - }, + usage: priced(28640, 8750, 720_000), timing: fabro_types::StageTiming::wall_only(154_000), started_at: None, state: Some(StageState::Succeeded), }, - RunBillingStage { - stage: BillingStageRef { + RunUsageStage { + stage: UsageStageRef { id: "review-changes".into(), name: "Review Changes".into(), }, - model: Some(billing_model( + model: Some(usage_model( lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex", )), - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 9120, - output_tokens: 2640, - reasoning_tokens: 0, - total_tokens: 11760, - total_usd_micros: Some(190_000), - }, + usage: priced(9120, 2640, 190_000), timing: fabro_types::StageTiming::wall_only(45_000), started_at: None, state: Some(StageState::Succeeded), }, - RunBillingStage { - stage: BillingStageRef { + RunUsageStage { + stage: UsageStageRef { id: "apply-changes".into(), name: "Apply Changes".into(), }, - model: Some(billing_model( + model: Some(usage_model( lithos_llm::catalog::builtin::anthropic(), "claude-opus-4-6", )), - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 21300, - output_tokens: 6480, - reasoning_tokens: 0, - total_tokens: 27780, - total_usd_micros: Some(870_000), - }, + usage: priced(21300, 6480, 870_000), timing: fabro_types::StageTiming::wall_only(118_000), started_at: None, state: Some(StageState::Running), }, ], - totals: RunBillingTotals { - cache_read_tokens: 0, - cache_write_tokens: 0, - timing: fabro_types::RunTiming::wall_only(389_000), - input_tokens: 71540, - output_tokens: 21080, - reasoning_tokens: 0, - total_tokens: 92620, - total_usd_micros: Some(2_260_000), + totals: RunUsageTotals { + timing: fabro_types::RunTiming::wall_only(389_000), + usage: priced(71540, 21080, 2_260_000), }, by_model: vec![ - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 33780, - output_tokens: 9690, - reasoning_tokens: 0, - total_tokens: 43470, - total_usd_micros: Some(1_350_000), - }, - model: billing_model( + UsageByModel { + usage: priced(33780, 9690, 1_350_000), + model: usage_model( lithos_llm::catalog::builtin::anthropic(), "claude-opus-4-6", ), - stages: 2, + stages: 2, }, - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 28640, - output_tokens: 8750, - reasoning_tokens: 0, - total_tokens: 37390, - total_usd_micros: Some(720_000), - }, - model: billing_model( + UsageByModel { + usage: priced(28640, 8750, 720_000), + model: usage_model( lithos_llm::catalog::builtin::gemini(), "gemini-3.1-pro-preview", ), - stages: 1, + stages: 1, }, - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 9120, - output_tokens: 2640, - reasoning_tokens: 0, - total_tokens: 11760, - total_usd_micros: Some(190_000), - }, - model: billing_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"), - stages: 1, + UsageByModel { + usage: priced(9120, 2640, 190_000), + model: usage_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"), + stages: 1, }, ], } @@ -2243,76 +2196,62 @@ mod workflows { } } -mod billing { +mod usage { use fabro_api::types::*; use lithos_llm::catalog::ProviderId; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; - fn billing_model(provider: ProviderId, model_id: &str) -> BillingModelRef { - BillingModelRef { + fn usage_model(provider: ProviderId, model_id: &str) -> UsageModelRef { + UsageModelRef { provider, model_id: model_id.into(), speed: None, } } - pub(super) fn aggregate() -> AggregateBilling { - AggregateBilling { - totals: AggregateBillingTotals { - cache_read_tokens: 0, - cache_write_tokens: 0, - runs: 9, - input_tokens: 643_860, - output_tokens: 189_720, - reasoning_tokens: 0, - timing: fabro_types::RunTiming::wall_only(3_501_000), - total_tokens: 833_580, - total_usd_micros: Some(20_340_000), + /// Demo usage priced from the catalog. + fn priced(input: u64, output: u64, usd_micros: u64) -> Usage { + Usage { + tokens: TokenCounts { + input, + output, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), + } + } + + pub(super) fn aggregate() -> AggregateUsage { + AggregateUsage { + totals: AggregateUsageTotals { + runs: 9, + timing: fabro_types::RunTiming::wall_only(3_501_000), + usage: priced(643_860, 189_720, 20_340_000), }, by_model: vec![ - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 304_020, - output_tokens: 87_210, - reasoning_tokens: 0, - total_tokens: 391_230, - total_usd_micros: Some(12_150_000), - }, - model: billing_model( + UsageByModel { + usage: priced(304_020, 87_210, 12_150_000), + model: usage_model( lithos_llm::catalog::builtin::anthropic(), "claude-opus-4-6", ), - stages: 18, + stages: 18, }, - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 257_760, - output_tokens: 78_750, - reasoning_tokens: 0, - total_tokens: 336_510, - total_usd_micros: Some(6_480_000), - }, - model: billing_model( + UsageByModel { + usage: priced(257_760, 78_750, 6_480_000), + model: usage_model( lithos_llm::catalog::builtin::gemini(), "gemini-3.1-pro-preview", ), - stages: 9, + stages: 9, }, - BillingByModel { - billing: BilledTokenCounts { - cache_read_tokens: 0, - cache_write_tokens: 0, - input_tokens: 82_080, - output_tokens: 23_760, - reasoning_tokens: 0, - total_tokens: 105_840, - total_usd_micros: Some(1_710_000), - }, - model: billing_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"), - stages: 9, + UsageByModel { + usage: priced(82_080, 23_760, 1_710_000), + model: usage_model(lithos_llm::catalog::builtin::openai(), "gpt-5.3-codex"), + stages: 9, }, ], } diff --git a/lib/apps/fabro-server/src/run_files.rs b/lib/apps/fabro-server/src/run_files.rs index d4d87a015..8c64da055 100644 --- a/lib/apps/fabro-server/src/run_files.rs +++ b/lib/apps/fabro-server/src/run_files.rs @@ -2308,7 +2308,7 @@ index 1111111..2222222 160000 failure: None, final_git_commit_sha: None, stages: Vec::new(), - billing: None, + usage: None, total_retries: 0, diff: fabro_types::RunDiff { patch: Some(patch.to_string()), diff --git a/lib/apps/fabro-server/src/run_tool_create.rs b/lib/apps/fabro-server/src/run_tool_create.rs index 503236f61..01d7ae7df 100644 --- a/lib/apps/fabro-server/src/run_tool_create.rs +++ b/lib/apps/fabro-server/src/run_tool_create.rs @@ -12,6 +12,7 @@ use fabro_types::{ }; use httpmock::Method::{GET, POST}; use httpmock::{HttpMockRequest, HttpMockResponse, MockServer}; +use lithos_llm::types::Usage; use serde_json::json; use tokio::fs; #[expect( @@ -216,7 +217,7 @@ fn run_with_status( completed_at: None, }, timing: None, - billing: None, + usage: Usage::default(), size: fabro_types::RunSize::default(), ask_fabro: fabro_types::AskFabro::default(), diff: None, diff --git a/lib/apps/fabro-server/src/server.rs b/lib/apps/fabro-server/src/server.rs index 83e409751..7c3a7c83e 100644 --- a/lib/apps/fabro-server/src/server.rs +++ b/lib/apps/fabro-server/src/server.rs @@ -23,30 +23,30 @@ use base64::engine::general_purpose::STANDARD as BASE64_STANDARD; use bytes::Bytes; use chrono::{DateTime, Utc}; pub use fabro_api::types::{ - AggregateBilling, AggregateBillingTotals, ApiQuestion, AppendEventResponse, ArtifactEntry, + AggregateUsage, AggregateUsageTotals, ApiQuestion, AppendEventResponse, ArtifactEntry, ArtifactListResponse, BatchDeleteRunsRequest, BatchDeleteRunsResponse, BatchDeleteRunsResult, BatchDeleteRunsResultOutcome, BatchDeleteRunsSummary, BatchRunLifecycleRequest, BatchRunLifecycleResponse, BatchRunLifecycleResult, BatchRunLifecycleResultOutcome, - BatchRunLifecycleSummary, BillingByModel, BillingStageRef, CloseRunPullRequestResponse, - CompletionResponse, CompletionUsage, CreateCompletionRequest, CreateRunPullRequestRequest, - CreateSecretRequest, CreateVariableRequest, DeleteRunResponse, DeleteRunSandbox, - DeleteSecretRequest, DenyRunRequest, DiskUsageResponse, DiskUsageRunRow, DiskUsageSummaryRow, - ErrorResponseEntry, ForkRequest, ForkResponse, IntegrationConnectionKind, - IntegrationConnectionState, IntegrationConnectionStatus, IntegrationProvider, - IntegrationStatus, LinkRunPullRequestRequest, MergeRunPullRequestRequest, - MergeRunPullRequestResponse, ModelReference, PaginatedEventList, PaginatedRunList, - PaginationMeta, PreflightResponse, PreviewUrlRequest, PreviewUrlResponse, Provider, - ProviderCredentialTestRequest, ProviderCredentialTestResponse, ProviderList, PruneRunEntry, - PruneRunsRequest, PruneRunsResponse, RenderWorkflowGraphDirection, RenderWorkflowGraphRequest, - RewindRequest, RewindResponse, Run, RunArtifactEntry, RunArtifactListResponse, RunBilling, - RunBillingStage, RunBillingTotals, RunError, RunManifest, RunStage, SandboxDetails, - SandboxFileEntry, SandboxFileListResponse, SandboxService, SandboxServiceListResponse, - SshAccessRequest, SshAccessResponse, StageHandler, StageState, StartRunRequest, - SubmitAnswerRequest, SystemCpuResourceScope, SystemCpuResources, SystemDiskResourceScope, - SystemDiskResources, SystemInfoResponse, SystemIntegrationStatus, SystemIntegrationsResponse, - SystemMemoryResourceScope, SystemMemoryResources, SystemRepairRunIssue, - SystemRepairRunsResponse, SystemResourcesResponse, SystemRunCounts, TimelineEntryResponse, - UpdateVariableRequest, VariableListResponse, VncPreviewResponse, WriteBlobResponse, + BatchRunLifecycleSummary, CloseRunPullRequestResponse, CompletionResponse, + CreateCompletionRequest, CreateRunPullRequestRequest, CreateSecretRequest, + CreateVariableRequest, DeleteRunResponse, DeleteRunSandbox, DeleteSecretRequest, + DenyRunRequest, DiskUsageResponse, DiskUsageRunRow, DiskUsageSummaryRow, ErrorResponseEntry, + ForkRequest, ForkResponse, IntegrationConnectionKind, IntegrationConnectionState, + IntegrationConnectionStatus, IntegrationProvider, IntegrationStatus, LinkRunPullRequestRequest, + MergeRunPullRequestRequest, MergeRunPullRequestResponse, ModelReference, PaginatedEventList, + PaginatedRunList, PaginationMeta, PreflightResponse, PreviewUrlRequest, PreviewUrlResponse, + Provider, ProviderCredentialTestRequest, ProviderCredentialTestResponse, ProviderList, + PruneRunEntry, PruneRunsRequest, PruneRunsResponse, RenderWorkflowGraphDirection, + RenderWorkflowGraphRequest, RewindRequest, RewindResponse, Run, RunArtifactEntry, + RunArtifactListResponse, RunError, RunManifest, RunStage, RunUsage, RunUsageStage, + RunUsageTotals, SandboxDetails, SandboxFileEntry, SandboxFileListResponse, SandboxService, + SandboxServiceListResponse, SshAccessRequest, SshAccessResponse, StageHandler, StageState, + StartRunRequest, SubmitAnswerRequest, SystemCpuResourceScope, SystemCpuResources, + SystemDiskResourceScope, SystemDiskResources, SystemInfoResponse, SystemIntegrationStatus, + SystemIntegrationsResponse, SystemMemoryResourceScope, SystemMemoryResources, + SystemRepairRunIssue, SystemRepairRunsResponse, SystemResourcesResponse, SystemRunCounts, + TimelineEntryResponse, UpdateVariableRequest, UsageByModel, UsageStageRef, + VariableListResponse, VncPreviewResponse, WriteBlobResponse, }; use fabro_auth::SqlVaultCredentialSource; use fabro_automation::{self, AutomationStore}; @@ -88,7 +88,7 @@ use fabro_types::settings::server::{ GithubIntegrationSettings, GithubIntegrationStrategy, LogDestination, }; use fabro_types::{ - AgentBackend, AskFabro, AskFabroUnavailableReason, BilledTokenCounts, BlobHash, EventBody, + AgentBackend, AskFabro, AskFabroUnavailableReason, BlobHash, EventBody, InterviewQuestionRecord, ModelRef, ModelTestMode, PairId, PairMessageId, PairTarget, PendingReason, Principal, PullRequestLink, QuestionType, RunControlAction, RunEvent, RunId, RunRunnableSource, RunStatusKind, SandboxProviderKind, ServerSettings, SessionCapability, @@ -113,6 +113,7 @@ use fabro_workflow::run_status::{FailureReason, RunStatus, SuccessReason}; use fabro_workflow::{Error as WorkflowError, operations, pull_request}; use futures_util::future::join_all; use lithos_llm::catalog::ProviderId; +use lithos_llm::types::Usage; use sha2::{Digest, Sha256}; use tempfile::NamedTempFile; use tokio::fs; @@ -317,19 +318,19 @@ enum ExecutionResult { const WORKER_CANCEL_GRACE: Duration = Duration::from_secs(5); const TERMINAL_DELETE_WORKER_GRACE: Duration = Duration::from_millis(50); const WORKER_CONTROL_ENQUEUE_TIMEOUT: Duration = Duration::from_secs(1); -/// Per-model billing totals. +/// Per-model usage totals. #[derive(Default)] -struct ModelBillingTotals { - stages: i64, - billing: BilledTokenCounts, +pub(crate) struct ModelUsageTotals { + pub(crate) stages: i64, + pub(crate) usage: Usage, } -/// In-memory aggregate billing counters, reset on server restart. +/// In-memory aggregate usage counters, reset on server restart. #[derive(Default)] -struct BillingAccumulator { - total_runs: i64, - total_timing: fabro_types::RunTiming, - by_model: HashMap, +pub(crate) struct UsageAccumulator { + pub(crate) total_runs: i64, + pub(crate) total_timing: fabro_types::RunTiming, + pub(crate) by_model: HashMap, } pub(crate) type RegistryFactoryOverride = @@ -1098,7 +1099,7 @@ fn resolve_slack_lifecycle_route_channel( /// Shared application state for the server. pub struct AppState { runs: Mutex>, - aggregate_billing: Mutex, + aggregate_usage: Mutex, pub(crate) stores: AppStores, session_runtimes: SessionRuntimeManager, artifact_store: ArtifactStore, @@ -1296,16 +1297,16 @@ pub(crate) struct ResolvedAppStateSettings { pub(crate) llm_overlay: LlmLayer, } -fn accumulate_billing_rollup( - accumulator: &mut BillingAccumulator, - rollup: &fabro_workflow::ProjectionBillingRollup, +fn accumulate_usage_rollup( + accumulator: &mut UsageAccumulator, + rollup: &fabro_workflow::ProjectionUsageRollup, ) { accumulator.total_runs += 1; accumulator.total_timing = accumulator.total_timing.saturating_add(&rollup.timing); for model in &rollup.by_model { let entry = accumulator.by_model.entry(model.model.clone()).or_default(); entry.stages += model.stages; - entry.billing.add_counts(&model.billing); + entry.usage = entry.usage.saturating_add(model.usage); } } @@ -2558,7 +2559,7 @@ pub(crate) fn build_app_state(config: AppStateConfig) -> anyhow::Result, run_id: RunId) { if let Some(ref projection) = final_projection { if projection.current_checkpoint().is_some() { let mut agg = state - .aggregate_billing + .aggregate_usage .lock() - .expect("aggregate_billing lock poisoned"); - accumulate_billing_rollup( + .expect("aggregate_usage lock poisoned"); + accumulate_usage_rollup( &mut agg, - &fabro_workflow::billing_rollup_from_projection(projection), + &fabro_workflow::usage_rollup_from_projection(projection), ); } } @@ -4473,12 +4474,12 @@ async fn execute_run_subprocess(state: Arc, run_id: RunId) { if final_state.current_checkpoint().is_some() { let mut agg = state - .aggregate_billing + .aggregate_usage .lock() - .expect("aggregate_billing lock poisoned"); - accumulate_billing_rollup( + .expect("aggregate_usage lock poisoned"); + accumulate_usage_rollup( &mut agg, - &fabro_workflow::billing_rollup_from_projection(&final_state), + &fabro_workflow::usage_rollup_from_projection(&final_state), ); } diff --git a/lib/apps/fabro-server/src/server/handler/mod.rs b/lib/apps/fabro-server/src/server/handler/mod.rs index f9278d01f..c5d1c2018 100644 --- a/lib/apps/fabro-server/src/server/handler/mod.rs +++ b/lib/apps/fabro-server/src/server/handler/mod.rs @@ -9,7 +9,6 @@ use super::{ApiError, AppState, IntoResponse, Json, Response, StatusCode, demo}; mod artifacts; pub(in crate::server) mod automations; -mod billing; mod completions; mod environments; pub(in crate::server) mod events; @@ -27,6 +26,7 @@ mod secrets; mod sessions; mod steer; pub(in crate::server) mod system; +mod usage; mod variables; mod worker_control; mod workflow_versions; @@ -135,7 +135,7 @@ pub(super) fn demo_routes() -> Router> { "/runs/{id}/stages/{stageId}/artifacts/download", get(not_implemented), ) - .route("/runs/{id}/billing", get(demo::get_run_billing)) + .route("/runs/{id}/usage", get(demo::get_run_usage)) .route("/runs/{id}/settings", get(demo::get_run_settings)) .route("/runs/{id}/preview", post(demo::generate_preview_url_stub)) .route("/runs/{id}/ssh", post(demo::create_ssh_access_stub)) @@ -178,7 +178,7 @@ pub(super) fn demo_routes() -> Router> { .route("/system/df", get(demo::get_system_disk_usage)) .route("/system/repair/runs", get(demo::get_system_repair_runs)) .route("/system/prune/runs", post(demo::prune_runs)) - .route("/billing", get(demo::get_aggregate_billing)) + .route("/usage", get(demo::get_aggregate_usage)) .route("/workflows", get(demo::list_workflows)) .route("/workflows/{name}", get(demo::get_workflow)) .route("/workflows/{name}/runs", get(demo::list_workflow_runs)) @@ -208,7 +208,7 @@ pub(super) fn real_routes() -> Router> { .route("/insights/history", get(not_implemented)) .merge(runs::routes()) .merge(events::routes()) - .merge(billing::routes()) + .merge(usage::routes()) .merge(pull_requests::routes()) .merge(artifacts::routes()) .merge(automations::routes()) diff --git a/lib/apps/fabro-server/src/server/handler/pair.rs b/lib/apps/fabro-server/src/server/handler/pair.rs index 95fe1decb..d35fdd844 100644 --- a/lib/apps/fabro-server/src/server/handler/pair.rs +++ b/lib/apps/fabro-server/src/server/handler/pair.rs @@ -873,7 +873,7 @@ mod tests { fixtures, test_support, }; use fabro_workflow::event as workflow_event; - use pebble_coding_agent::events::{CodingAgentEvent, TokenUsage}; + use pebble_coding_agent::events::{CodingAgentEvent, Usage}; use tower::ServiceExt; use super::*; @@ -908,9 +908,7 @@ mod tests { CodingEvent::AssistantMessage { text: "I found the issue.".to_string(), model: "gpt-5.4".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window: None, reasoning: None, @@ -945,9 +943,7 @@ mod tests { CodingEvent::AssistantMessage { text: "wrong stage".to_string(), model: "gpt-5.4".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window: None, reasoning: None, diff --git a/lib/apps/fabro-server/src/server/handler/system.rs b/lib/apps/fabro-server/src/server/handler/system.rs index 8ab5e668b..dcd4bf5e5 100644 --- a/lib/apps/fabro-server/src/server/handler/system.rs +++ b/lib/apps/fabro-server/src/server/handler/system.rs @@ -10,18 +10,19 @@ use fabro_slack::config::{ }; use fabro_static::EnvVars; use fabro_types::settings::server::GithubIntegrationSettings; +use fabro_types::sum_usage; use fabro_vault::Vault; use tokio::time::timeout; use super::super::{ - AggregateBilling, AggregateBillingTotals, ApiError, AppState, BilledTokenCounts, - BillingByModel, DfParams, FABRO_VERSION, GithubIntegrationStrategy, IntegrationConnectionState, - IntegrationProvider, IntegrationStatus, IntoResponse, Json, Path, PruneRunsRequest, - PruneRunsResponse, Query, RequiredUser, Response, Router, RunStatus, State, StatusCode, - SystemInfoResponse, SystemIntegrationStatus, SystemIntegrationsResponse, SystemRepairRunIssue, - SystemRepairRunsResponse, SystemRunCounts, build_disk_usage_response, build_prune_plan, - counts_toward_scheduler_capacity, delete_run_internal, diagnostics, get, post, - resource_sampler, spawn_blocking, system_sandbox_provider, to_i64, + AggregateUsage, AggregateUsageTotals, ApiError, AppState, DfParams, FABRO_VERSION, + GithubIntegrationStrategy, IntegrationConnectionState, IntegrationProvider, IntegrationStatus, + IntoResponse, Json, Path, PruneRunsRequest, PruneRunsResponse, Query, RequiredUser, Response, + Router, RunStatus, State, StatusCode, SystemInfoResponse, SystemIntegrationStatus, + SystemIntegrationsResponse, SystemRepairRunIssue, SystemRepairRunsResponse, SystemRunCounts, + UsageByModel, build_disk_usage_response, build_prune_plan, counts_toward_scheduler_capacity, + delete_run_internal, diagnostics, get, post, resource_sampler, spawn_blocking, + system_sandbox_provider, to_i64, }; const SERVER_DIAGNOSTICS_TIMEOUT: Duration = Duration::from_secs(25); @@ -38,7 +39,7 @@ pub(super) fn routes() -> Router> { .route("/system/df", get(get_system_df)) .route("/system/repair/runs", get(get_system_repair_runs)) .route("/system/prune/runs", post(prune_runs)) - .route("/billing", get(get_aggregate_billing)) + .route("/usage", get(get_aggregate_usage)) } pub(in crate::server) async fn health() -> Response { @@ -715,41 +716,26 @@ pub(in crate::server) async fn openapi_spec() -> Response { Json(value).into_response() } -async fn get_aggregate_billing( - _auth: RequiredUser, - State(state): State>, -) -> Response { +async fn get_aggregate_usage(_auth: RequiredUser, State(state): State>) -> Response { let agg = state - .aggregate_billing + .aggregate_usage .lock() - .expect("aggregate_billing lock poisoned"); - let by_model: Vec = agg + .expect("aggregate_usage lock poisoned"); + let by_model: Vec = agg .by_model .iter() - .map(|(model, totals)| BillingByModel { - billing: totals.billing.clone(), - model: model.clone(), - stages: totals.stages, + .map(|(model, totals)| UsageByModel { + model: model.clone(), + stages: totals.stages, + usage: totals.usage, }) .collect(); - let total_billing = - agg.by_model - .values() - .fold(BilledTokenCounts::default(), |mut acc, totals| { - acc.add_counts(&totals.billing); - acc - }); - let response = AggregateBilling { - totals: AggregateBillingTotals { - cache_read_tokens: total_billing.cache_read_tokens, - cache_write_tokens: total_billing.cache_write_tokens, - input_tokens: total_billing.input_tokens, - output_tokens: total_billing.output_tokens, - reasoning_tokens: total_billing.reasoning_tokens, - runs: agg.total_runs, - timing: agg.total_timing, - total_tokens: total_billing.total_tokens, - total_usd_micros: total_billing.total_usd_micros, + let usage = sum_usage(agg.by_model.values().map(|totals| totals.usage)); + let response = AggregateUsage { + totals: AggregateUsageTotals { + runs: agg.total_runs, + timing: agg.total_timing, + usage, }, by_model, }; diff --git a/lib/apps/fabro-server/src/server/handler/billing.rs b/lib/apps/fabro-server/src/server/handler/usage.rs similarity index 72% rename from lib/apps/fabro-server/src/server/handler/billing.rs rename to lib/apps/fabro-server/src/server/handler/usage.rs index aea19a325..e92082e6b 100644 --- a/lib/apps/fabro-server/src/server/handler/billing.rs +++ b/lib/apps/fabro-server/src/server/handler/usage.rs @@ -4,18 +4,19 @@ use std::sync::Arc; use chrono::{DateTime, Utc}; use fabro_types::{ Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming, + usage_is_empty, }; use super::super::{ - AppState, BillingByModel, BillingStageRef, IntoResponse, Json, ListResponse, PaginationParams, - Path, Query, RequiredUser, Response, Router, RunBilling, RunBillingStage, RunBillingTotals, - RunId, RunStage, State, StatusCode, get, parse_run_id_path, + AppState, IntoResponse, Json, ListResponse, PaginationParams, Path, Query, RequiredUser, + Response, Router, RunId, RunStage, RunUsage, RunUsageStage, RunUsageTotals, State, StatusCode, + UsageByModel, UsageStageRef, get, parse_run_id_path, }; pub(super) fn routes() -> Router> { Router::new() .route("/runs/{id}/stages", get(list_run_stages)) - .route("/runs/{id}/billing", get(get_run_billing)) + .route("/runs/{id}/usage", get(get_run_usage)) } fn run_stage_from_projection( @@ -41,7 +42,7 @@ fn run_stage_from_projection( id: stage_id.clone(), name: stage_id.node_id().to_owned(), handler, - billing: stage.usage.clone(), + usage: stage.usage, status: stage.effective_state(), wall_time_ms: stage.live_wall_time_ms(now), node_id: stage_id.node_id().to_owned(), @@ -82,7 +83,7 @@ async fn list_run_stages( (StatusCode::OK, Json(ListResponse::new(stages))).into_response() } -async fn get_run_billing( +async fn get_run_usage( _auth: RequiredUser, State(state): State>, Path(id): Path, @@ -92,14 +93,14 @@ async fn get_run_billing( Err(err) => return err.into_response(), }; - let rollup = fabro_workflow::billing_rollup_from_projection(&projection); + let rollup = fabro_workflow::usage_rollup_from_projection(&projection); let by_model = rollup .by_model .iter() - .map(|model| BillingByModel { - billing: model.billing.clone(), - model: model.model.clone(), - stages: model.stages, + .map(|model| UsageByModel { + model: model.model.clone(), + stages: model.stages, + usage: model.usage, }) .collect::>(); @@ -108,7 +109,7 @@ async fn get_run_billing( .iter() .map(|stage| (stage.node_id.as_str(), stage)) .collect::>(); - let live_rows = live_billing_rows(&projection, Utc::now()); + let live_rows = live_usage_rows(&projection, Utc::now()); let totals_timing = live_rows.iter().fold(StageTiming::default(), |acc, row| { acc.saturating_add(&row.timing) }); @@ -116,13 +117,11 @@ async fn get_run_billing( .into_iter() .map(|row| { let rollup_stage = rollup_by_node.get(row.node_id.as_str()); - RunBillingStage { - billing: rollup_stage - .map(|stage| stage.billing.clone()) - .unwrap_or_default(), + RunUsageStage { + usage: rollup_stage.map(|stage| stage.usage).unwrap_or_default(), model: rollup_stage.and_then(|stage| stage.model.as_ref()).cloned(), timing: row.timing, - stage: BillingStageRef { + stage: UsageStageRef { id: row.node_id.clone(), name: row.node_id, }, @@ -132,25 +131,19 @@ async fn get_run_billing( }) .collect::>(); - let response = RunBilling { + let response = RunUsage { by_model, stages, - totals: RunBillingTotals { - cache_read_tokens: rollup.totals.cache_read_tokens, - cache_write_tokens: rollup.totals.cache_write_tokens, - input_tokens: rollup.totals.input_tokens, - output_tokens: rollup.totals.output_tokens, - reasoning_tokens: rollup.totals.reasoning_tokens, - timing: totals_timing.into(), - total_tokens: rollup.totals.total_tokens, - total_usd_micros: rollup.totals.total_usd_micros, + totals: RunUsageTotals { + timing: totals_timing.into(), + usage: rollup.totals, }, }; (StatusCode::OK, Json(response)).into_response() } -struct LiveBillingRow { +struct LiveUsageRow { node_id: String, timing: StageTiming, started_at: Option>, @@ -158,19 +151,19 @@ struct LiveBillingRow { latest_visit: u32, } -fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec { +fn live_usage_rows(projection: &RunProjection, now: DateTime) -> Vec { let mut row_indices = HashMap::::new(); - let mut rows = Vec::::new(); + let mut rows = Vec::::new(); for (stage_id, stage) in projection.iter_stages() { let node_id = stage_id.node_id(); - if projection.is_boundary_stage(node_id) || !stage_has_billing_row(stage) { + if projection.is_boundary_stage(node_id) || !stage_has_usage_row(stage) { continue; } let index = *row_indices.entry(node_id.to_string()).or_insert_with(|| { let index = rows.len(); - rows.push(LiveBillingRow { + rows.push(LiveUsageRow { node_id: node_id.to_string(), timing: StageTiming::default(), started_at: None, @@ -193,9 +186,9 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime) -> Vec bool { +fn stage_has_usage_row(stage: &StageProjection) -> bool { stage.completion.is_some() || stage.timing.is_some() - || !stage.usage.is_zero() + || !usage_is_empty(&stage.usage) || stage.started_at.is_some() } diff --git a/lib/apps/fabro-server/src/server/tests.rs b/lib/apps/fabro-server/src/server/tests.rs index 066c9de5a..678ba7c61 100644 --- a/lib/apps/fabro-server/src/server/tests.rs +++ b/lib/apps/fabro-server/src/server/tests.rs @@ -37,9 +37,9 @@ use httpmock::Method::{GET, POST}; use httpmock::MockServer; use lithos_llm::catalog::ModelId; use lithos_llm::types::{ - ReasoningEffort, ReasoningOutput, Request as LlmRequest, Speed, TokenCounts, + Cost, CostSource, ReasoningEffort, ReasoningOutput, Request as LlmRequest, Speed, TokenCounts, }; -use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, TokenUsage}; +use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, Usage}; use serde_json::json; use tokio::sync::Notify; use tokio_stream::StreamExt as _; @@ -5681,8 +5681,8 @@ fn stage_completed_event(node_id: &str) -> workflow_event::Event { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -5714,9 +5714,7 @@ fn agent_message_event( CodingEvent::AssistantMessage { text: text.to_string(), model: "gpt-5.4".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window, reasoning, @@ -6017,11 +6015,10 @@ channel = "#deploys" artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ) .await; @@ -6088,7 +6085,7 @@ channel = "#deploys" final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ) .await; @@ -6252,11 +6249,10 @@ channel = "#deploys" artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ) .await; @@ -6369,11 +6365,10 @@ async fn persist_cancelled_run_status_ignores_already_terminal_runs() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -6405,11 +6400,10 @@ async fn delete_terminal_managed_run_does_not_send_cancel_signal() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -6528,8 +6522,8 @@ async fn list_run_stages_projects_retrying_until_completion() { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -6568,15 +6562,15 @@ async fn list_run_stages_projects_retrying_until_completion() { "work", 1, &workflow_event::Event::StageFailed { - node_id: "work".to_string(), - name: "Work".to_string(), - index: 1, - failure: FailureDetail::new("try again", FailureCategory::TransientInfra), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(10), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "work".to_string(), + name: "Work".to_string(), + index: 1, + failure: FailureDetail::new("try again", FailureCategory::TransientInfra), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(10), + usage_by_model: Vec::new(), + usage: None, + actor: None, }, ) .await; @@ -6624,8 +6618,8 @@ async fn list_run_stages_projects_retrying_until_completion() { status: "partially_succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -6700,7 +6694,7 @@ async fn list_run_stages_projects_running_stage_as_cancelled_after_cancelled_run final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ) .await @@ -6801,28 +6795,31 @@ async fn list_run_stages_includes_stage_model_usage() { ); } -fn test_billed_usage( +fn test_priced_usage( model_id: &str, input_tokens: u64, output_tokens: u64, -) -> fabro_types::BilledModelUsage { - let mut usage = fabro_types::BilledModelUsage::new( +) -> fabro_types::ModelUsage { + fabro_types::ModelUsage::new( ModelRef::new( lithos_llm::catalog::builtin::openai(), ModelId::new(model_id), ), - TokenCounts { - input: input_tokens, - output: output_tokens, - ..TokenCounts::default() + Usage { + tokens: TokenCounts { + input: input_tokens, + output: output_tokens, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: input_tokens + output_tokens, + source: CostSource::Catalog, + }), }, - None, - ); - usage.total_usd_micros = Some(i64::try_from(input_tokens + output_tokens).unwrap()); - usage + ) } -async fn create_billed_retry_run(state: &Arc, run_id: RunId) { +async fn create_priced_retry_run(state: &Arc, run_id: RunId) { create_durable_run_with_events(state, run_id, &[ workflow_event::Event::RunSubmitted { definition_blob: None, @@ -6838,15 +6835,15 @@ async fn create_billed_retry_run(state: &Arc, run_id: RunId) { "verify", 1, &workflow_event::Event::StageFailed { - node_id: "verify".to_string(), - name: "Verify".to_string(), - index: 1, - failure: FailureDetail::new("try again", FailureCategory::TransientInfra), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(1200), - billing_by_model: Vec::new(), - billing: Some(test_billed_usage("gpt-old", 100, 10)), - actor: None, + node_id: "verify".to_string(), + name: "Verify".to_string(), + index: 1, + failure: FailureDetail::new("try again", FailureCategory::TransientInfra), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(1200), + usage_by_model: Vec::new(), + usage: Some(test_priced_usage("gpt-old", 100, 10)), + actor: None, }, ) .await; @@ -6863,8 +6860,8 @@ async fn create_billed_retry_run(state: &Arc, run_id: RunId) { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: Some(test_billed_usage("gpt-new", 200, 20)), + usage_by_model: Vec::new(), + usage: Some(test_priced_usage("gpt-new", 200, 20)), failure: None, notes: None, files_touched: Vec::new(), @@ -6951,8 +6948,8 @@ async fn list_run_stages_distinguishes_visits() { status: "failed".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -7198,7 +7195,7 @@ async fn list_run_stages_exposes_parallel_branch_identity() { } #[tokio::test] -async fn run_billing_includes_live_stage_timing_in_rows_and_totals() { +async fn run_usage_includes_live_stage_timing_in_rows_and_totals() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); @@ -7225,7 +7222,7 @@ async fn run_billing_includes_live_stage_timing_in_rows_and_totals() { .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) @@ -7242,10 +7239,10 @@ async fn run_billing_includes_live_stage_timing_in_rows_and_totals() { } /// `checkpoint.completed_nodes` records every visit, so a looped node appears -/// once per re-entry. Billing must dedup so a retried node renders as one row +/// once per re-entry. Usage must dedup so a retried node renders as one row /// and `runtime_secs` is summed across all visits exactly once. #[tokio::test] -async fn run_billing_dedups_retried_nodes_and_sums_their_durations() { +async fn run_usage_dedups_retried_nodes_and_sums_their_durations() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); @@ -7273,8 +7270,8 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() { status: "failed".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -7305,8 +7302,8 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -7359,7 +7356,7 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() { .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) @@ -7390,14 +7387,14 @@ async fn run_billing_dedups_retried_nodes_and_sums_their_durations() { } #[tokio::test] -async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() { +async fn run_usage_sums_usage_across_retry_visits_and_uses_latest_model() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); - create_billed_retry_run(&state, run_id).await; - let success_usage = test_billed_usage("gpt-new", 200, 20); - let mut latest_outcome: Outcome> = Outcome::success(); + create_priced_retry_run(&state, run_id).await; + let success_usage = test_priced_usage("gpt-new", 200, 20); + let mut latest_outcome: Outcome> = Outcome::success(); latest_outcome.usage = Some(success_usage); latest_outcome.timing = Some(fabro_types::StageTiming::wall_only(800)); let run_store = state.stores.runs.open_run(&run_id).await.unwrap(); @@ -7433,7 +7430,7 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() { .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) @@ -7446,14 +7443,14 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() { assert_eq!(stages[0]["stage"]["id"], "verify"); assert_eq!(stages[0]["model"]["provider"], "openai"); assert_eq!(stages[0]["model"]["model_id"], "gpt-new"); - assert_eq!(stages[0]["billing"]["input_tokens"], 300); - assert_eq!(stages[0]["billing"]["output_tokens"], 30); - assert_eq!(stages[0]["billing"]["total_usd_micros"], 330); + assert_eq!(stages[0]["usage"]["tokens"]["input"], 300); + assert_eq!(stages[0]["usage"]["tokens"]["output"], 30); + assert_eq!(stages[0]["usage"]["cost"]["usd_micros"], 330); assert!(stages[0]["timing"]["wall_time_ms"].as_u64().unwrap() == 2000); - assert_eq!(body["totals"]["input_tokens"], 300); - assert_eq!(body["totals"]["output_tokens"], 30); - assert_eq!(body["totals"]["total_usd_micros"], 330); + assert_eq!(body["totals"]["usage"]["tokens"]["input"], 300); + assert_eq!(body["totals"]["usage"]["tokens"]["output"], 30); + assert_eq!(body["totals"]["usage"]["cost"]["usd_micros"], 330); assert!(body["totals"]["timing"]["wall_time_ms"].as_u64().unwrap() == 2000); let by_model = body["by_model"].as_array().unwrap(); @@ -7469,21 +7466,21 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() { assert_eq!(old_model["model"]["provider"], "openai"); assert_eq!(new_model["model"]["provider"], "openai"); assert_eq!(old_model["stages"], 1); - assert_eq!(old_model["billing"]["input_tokens"], 100); + assert_eq!(old_model["usage"]["tokens"]["input"], 100); assert_eq!(new_model["stages"], 1); - assert_eq!(new_model["billing"]["input_tokens"], 200); + assert_eq!(new_model["usage"]["tokens"]["input"], 200); } -/// The stage popover reads `billing` off the stages list, so it must be scoped -/// to one visit — unlike the Billing tab's rows, which sum every visit of a +/// The stage popover reads `usage` off the stages list, so it must be scoped +/// to one visit — unlike the Usage tab's rows, which sum every visit of a /// node. #[tokio::test] -async fn list_run_stages_reports_billing_per_visit() { +async fn list_run_stages_reports_usage_per_visit() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); - create_billed_retry_run(&state, run_id).await; + create_priced_retry_run(&state, run_id).await; let response = app .oneshot( @@ -7498,18 +7495,18 @@ async fn list_run_stages_reports_billing_per_visit() { let body = response_json!(response, StatusCode::OK).await; let first = stage_entry(&body, "verify@1"); - assert_eq!(first["billing"]["input_tokens"], 100); - assert_eq!(first["billing"]["output_tokens"], 10); - assert_eq!(first["billing"]["total_usd_micros"], 110); + assert_eq!(first["usage"]["tokens"]["input"], 100); + assert_eq!(first["usage"]["tokens"]["output"], 10); + assert_eq!(first["usage"]["cost"]["usd_micros"], 110); let second = stage_entry(&body, "verify@2"); - assert_eq!(second["billing"]["input_tokens"], 200); - assert_eq!(second["billing"]["output_tokens"], 20); - assert_eq!(second["billing"]["total_usd_micros"], 220); + assert_eq!(second["usage"]["tokens"]["input"], 200); + assert_eq!(second["usage"]["tokens"]["output"], 20); + assert_eq!(second["usage"]["cost"]["usd_micros"], 220); } #[tokio::test] -async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() { +async fn list_run_stages_reports_zero_usage_for_a_stage_that_called_no_model() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); @@ -7537,11 +7534,11 @@ async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() .unwrap(); let body = response_json!(response, StatusCode::OK).await; - let billing = &stage_entry(&body, "script@1")["billing"]; - assert_eq!(billing["input_tokens"], 0); - assert_eq!(billing["output_tokens"], 0); + let usage = &stage_entry(&body, "script@1")["usage"]; + assert_eq!(usage["tokens"]["input"], 0); + assert_eq!(usage["tokens"]["output"], 0); // No model ran, so there is nothing to price — not a $0.00 cost. - assert!(billing.get("total_usd_micros").is_none()); + assert!(usage.get("cost").is_none()); } #[tokio::test] @@ -7582,15 +7579,15 @@ async fn list_run_stages_shows_retrying_after_failed_event() { "work", 1, &workflow_event::Event::StageFailed { - node_id: "work".to_string(), - name: "Work".to_string(), - index: 0, - failure: FailureDetail::new("flake", FailureCategory::TransientInfra), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(5), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "work".to_string(), + name: "Work".to_string(), + index: 0, + failure: FailureDetail::new("flake", FailureCategory::TransientInfra), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(5), + usage_by_model: Vec::new(), + usage: None, + actor: None, }, ) .await; @@ -7665,15 +7662,15 @@ async fn list_run_stages_shows_retrying_when_failed_will_retry() { "work", 1, &workflow_event::Event::StageFailed { - node_id: "work".to_string(), - name: "Work".to_string(), - index: 0, - failure: FailureDetail::new("flake", FailureCategory::TransientInfra), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(5), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "work".to_string(), + name: "Work".to_string(), + index: 0, + failure: FailureDetail::new("flake", FailureCategory::TransientInfra), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(5), + usage_by_model: Vec::new(), + usage: None, + actor: None, }, ) .await; @@ -7694,7 +7691,7 @@ async fn list_run_stages_shows_retrying_when_failed_will_retry() { } #[tokio::test] -async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attempt_duration() { +async fn run_usage_retried_node_then_succeeded_emits_one_row_with_final_attempt_duration() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); @@ -7716,15 +7713,15 @@ async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attemp max_attempts: 3, }, workflow_event::Event::StageFailed { - node_id: "work".to_string(), - name: "Work".to_string(), - index: 0, - failure: FailureDetail::new("transient", FailureCategory::TransientInfra), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(10), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "work".to_string(), + name: "Work".to_string(), + index: 0, + failure: FailureDetail::new("transient", FailureCategory::TransientInfra), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(10), + usage_by_model: Vec::new(), + usage: None, + actor: None, }, workflow_event::Event::StageRetrying { node_id: "work".to_string(), @@ -7752,8 +7749,8 @@ async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attemp status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -7774,7 +7771,7 @@ async fn run_billing_retried_node_then_succeeded_emits_one_row_with_final_attemp .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) @@ -7824,8 +7821,8 @@ fn revisit_test_completed_with_visit( status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -7842,7 +7839,7 @@ fn revisit_test_completed_with_visit( } #[tokio::test] -async fn run_billing_revisited_node_collapses_to_two_rows_with_summed_visit_duration() { +async fn run_usage_revisited_node_collapses_to_two_rows_with_summed_visit_duration() { let state = test_app_state_with_isolated_storage(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let run_id = RunId::new(); @@ -7868,7 +7865,7 @@ async fn run_billing_revisited_node_collapses_to_two_rows_with_summed_visit_dura .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) @@ -7942,11 +7939,10 @@ async fn create_unreadable_durable_run(state: &Arc, run_id: RunId) { artifact_count: 0, status: "legacy-status".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, "2026-05-05T20:46:33Z".parse().unwrap(), None, @@ -8280,11 +8276,10 @@ async fn create_completed_run_ready_for_pull_request( artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: Some("final-sha".to_string()), final_patch: Some(final_patch.to_string()), diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -11889,18 +11884,18 @@ async fn run_projection_endpoints_reflect_events_appended_to_an_open_run() { let checkpoint = response_json!(checkpoint, StatusCode::OK).await; assert_eq!(checkpoint["git_commit_sha"].as_str(), Some("cache-sha")); - let billing = app + let usage = app .oneshot( Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) .unwrap(), ) .await .unwrap(); - let billing = response_json!(billing, StatusCode::OK).await; - assert_eq!(billing["stages"][0]["stage"]["id"].as_str(), Some("review")); + let usage = response_json!(usage, StatusCode::OK).await; + assert_eq!(usage["stages"][0]["stage"]["id"].as_str(), Some("review")); } #[tokio::test] @@ -13377,7 +13372,7 @@ async fn worker_token_is_rejected_on_user_only_routes() { Method::GET, format!("/runs/{run_id}/stages/code@2/artifacts/download"), ), - (Method::GET, format!("/runs/{run_id}/billing")), + (Method::GET, format!("/runs/{run_id}/usage")), (Method::GET, format!("/runs/{run_id}/settings")), (Method::POST, format!("/runs/{run_id}/preview")), (Method::POST, format!("/runs/{run_id}/ssh")), @@ -13854,11 +13849,10 @@ async fn patch_run_title_updates_active_and_archived_runs() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ) .await @@ -14138,11 +14132,10 @@ async fn retry_succeeded_run_creates_and_queues_new_run() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -14277,11 +14270,10 @@ async fn cancel_terminal_durable_run_returns_conflict() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -14327,11 +14319,10 @@ async fn steer_terminal_durable_run_returns_run_not_steerable() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -14768,8 +14759,8 @@ async fn active_acp_steerable_marker_clears_on_terminal_paths() { status: "success".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -14784,15 +14775,15 @@ async fn active_acp_steerable_marker_clears_on_terminal_paths() { max_attempts: 1, }, workflow_event::Event::StageFailed { - node_id: "agent".to_string(), - name: "agent".to_string(), - index: 0, - failure: FailureDetail::new("failed", FailureCategory::Deterministic), - will_retry: false, - timing: fabro_types::StageTiming::wall_only(1), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "agent".to_string(), + name: "agent".to_string(), + index: 0, + failure: FailureDetail::new("failed", FailureCategory::Deterministic), + will_retry: false, + timing: fabro_types::StageTiming::wall_only(1), + usage_by_model: Vec::new(), + usage: None, + actor: None, }, ]; @@ -15154,7 +15145,8 @@ async fn list_runs_returns_started_run() { assert!(run_json_status(&items[0]).is_object()); assert!(items[0]["labels"].is_object()); assert!(run_json_pending_control(&items[0]).is_null()); - assert!(items[0]["billing"].is_null()); + assert_eq!(items[0]["usage"]["tokens"]["input"], 0); + assert!(items[0]["usage"].get("cost").is_none()); } #[tokio::test] @@ -15174,11 +15166,10 @@ async fn archive_and_unarchive_updates_listing_visibility() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -15291,11 +15282,10 @@ fn workflow_completed_event() -> workflow_event::Event { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, } } @@ -16116,11 +16106,10 @@ async fn delete_run_retry_after_missing_provider_resource_removes_metadata() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -16225,54 +16214,51 @@ async fn delete_active_run_force_succeeds() { } #[tokio::test] -async fn get_aggregate_billing_returns_zeros_initially() { +async fn get_aggregate_usage_returns_zeros_initially() { let state = test_app_state(); let app = crate::test_support::build_test_router(Arc::clone(&state)); let req = Request::builder() .method("GET") - .uri(api("/billing")) + .uri(api("/usage")) .body(Body::empty()) .unwrap(); let response = app.oneshot(req).await.unwrap(); let body = response_json!(response, StatusCode::OK).await; assert_eq!(body["totals"]["runs"].as_i64().unwrap(), 0); - assert_eq!(body["totals"]["input_tokens"].as_i64().unwrap(), 0); - assert_eq!(body["totals"]["output_tokens"].as_i64().unwrap(), 0); + assert_eq!( + body["totals"]["usage"]["tokens"]["input"].as_u64().unwrap(), + 0 + ); + assert_eq!( + body["totals"]["usage"]["tokens"]["output"] + .as_u64() + .unwrap(), + 0 + ); assert_eq!( body["totals"]["timing"]["wall_time_ms"].as_u64().unwrap(), 0 ); - assert!(body["totals"]["total_usd_micros"].is_null()); + assert!(body["totals"]["usage"].get("cost").is_none()); assert!(body["by_model"].as_array().unwrap().is_empty()); } #[tokio::test] -async fn get_aggregate_billing_returns_provider_model_speed_identity() { +async fn get_aggregate_usage_returns_provider_model_speed_identity() { let state = test_app_state(); { - let mut agg = state - .aggregate_billing - .lock() - .expect("aggregate billing lock"); + let mut agg = state.aggregate_usage.lock().expect("aggregate usage lock"); agg.total_runs = 1; agg.by_model.insert( ModelRef::new( lithos_llm::catalog::builtin::anthropic(), ModelId::new("claude-opus-4-6"), ), - ModelBillingTotals { - stages: 1, - billing: BilledTokenCounts { - input_tokens: 10, - output_tokens: 1, - total_tokens: 11, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(11), - }, + ModelUsageTotals { + stages: 1, + usage: test_priced_usage("claude-opus-4-6", 10, 1).usage, }, ); agg.by_model.insert( @@ -16281,17 +16267,9 @@ async fn get_aggregate_billing_returns_provider_model_speed_identity() { ModelId::new("claude-opus-4-6"), ) .with_speed(Some(Speed::Fast)), - ModelBillingTotals { - stages: 1, - billing: BilledTokenCounts { - input_tokens: 20, - output_tokens: 2, - total_tokens: 22, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(22), - }, + ModelUsageTotals { + stages: 1, + usage: test_priced_usage("claude-opus-4-6", 20, 2).usage, }, ); } @@ -16301,7 +16279,7 @@ async fn get_aggregate_billing_returns_provider_model_speed_identity() { .oneshot( Request::builder() .method("GET") - .uri(api("/billing")) + .uri(api("/usage")) .body(Body::empty()) .unwrap(), ) @@ -16321,31 +16299,31 @@ async fn get_aggregate_billing_returns_provider_model_speed_identity() { .unwrap(); assert_eq!(standard["model"]["provider"], "anthropic"); assert_eq!(standard["model"]["model_id"], "claude-opus-4-6"); - assert_eq!(standard["billing"]["input_tokens"], 10); + assert_eq!(standard["usage"]["tokens"]["input"], 10); assert_eq!(fast["model"]["provider"], "anthropic"); assert_eq!(fast["model"]["model_id"], "claude-opus-4-6"); - assert_eq!(fast["billing"]["input_tokens"], 20); + assert_eq!(fast["usage"]["tokens"]["input"], 20); } #[tokio::test] -async fn get_aggregate_billing_saturates_total_cost_across_models() { +async fn get_aggregate_usage_saturates_total_cost_across_models() { let state = test_app_state(); { - let mut agg = state - .aggregate_billing - .lock() - .expect("aggregate billing lock"); - for (model_id, total_usd_micros) in [("maximum", i64::MAX), ("one", 1)] { + let mut agg = state.aggregate_usage.lock().expect("aggregate usage lock"); + for (model_id, usd_micros) in [("maximum", u64::MAX), ("one", 1)] { agg.by_model.insert( ModelRef::new( lithos_llm::catalog::builtin::openai(), ModelId::new(model_id), ), - ModelBillingTotals { - stages: 1, - billing: BilledTokenCounts { - total_usd_micros: Some(total_usd_micros), - ..BilledTokenCounts::default() + ModelUsageTotals { + stages: 1, + usage: Usage { + tokens: TokenCounts::default(), + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), }, }, ); @@ -16357,7 +16335,7 @@ async fn get_aggregate_billing_saturates_total_cost_across_models() { .oneshot( Request::builder() .method("GET") - .uri(api("/billing")) + .uri(api("/usage")) .body(Body::empty()) .unwrap(), ) @@ -16365,63 +16343,42 @@ async fn get_aggregate_billing_saturates_total_cost_across_models() { .unwrap(); let body = response_json!(response, StatusCode::OK).await; - assert_eq!(body["totals"]["total_usd_micros"].as_i64(), Some(i64::MAX)); + assert_eq!( + body["totals"]["usage"]["cost"]["usd_micros"].as_u64(), + Some(u64::MAX) + ); } #[test] -fn aggregate_billing_counts_projection_rollup_usage_visits() { - let mut accumulator = BillingAccumulator::default(); - let rollup = fabro_workflow::ProjectionBillingRollup { - stages: Vec::new(), - totals: BilledTokenCounts { - input_tokens: 300, - output_tokens: 30, - total_tokens: 330, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(330), - }, - by_model: vec![ - fabro_workflow::ProjectionBillingByModel { - model: ModelRef::new( +fn aggregate_usage_counts_projection_rollup_usage_visits() { + let mut accumulator = UsageAccumulator::default(); + let rollup = fabro_workflow::ProjectionUsageRollup { + stages: Vec::new(), + totals: test_priced_usage("gpt-5.4", 300, 30).usage, + by_model: vec![ + fabro_workflow::ProjectionUsageByModel { + model: ModelRef::new( lithos_llm::catalog::builtin::openai(), ModelId::new("gpt-5.4"), ), - stages: 1, - billing: BilledTokenCounts { - input_tokens: 100, - output_tokens: 10, - total_tokens: 110, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(110), - }, + stages: 1, + usage: test_priced_usage("gpt-5.4", 100, 10).usage, }, - fabro_workflow::ProjectionBillingByModel { - model: ModelRef::new( + fabro_workflow::ProjectionUsageByModel { + model: ModelRef::new( lithos_llm::catalog::builtin::openai(), ModelId::new("gpt-5.4"), ) .with_speed(Some(Speed::Fast)), - stages: 1, - billing: BilledTokenCounts { - input_tokens: 200, - output_tokens: 20, - total_tokens: 220, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: Some(220), - }, + stages: 1, + usage: test_priced_usage("gpt-5.4", 200, 20).usage, }, ], - timing: fabro_types::RunTiming::wall_only(2000), - billed_visit_count: 2, + timing: fabro_types::RunTiming::wall_only(2000), + usage_visit_count: 2, }; - accumulate_billing_rollup(&mut accumulator, &rollup); + accumulate_usage_rollup(&mut accumulator, &rollup); assert_eq!(accumulator.total_runs, 1); assert_eq!(accumulator.total_timing.wall_time_ms, 2000); @@ -16439,8 +16396,9 @@ fn aggregate_billing_counts_projection_rollup_usage_visits() { lithos_llm::catalog::builtin::openai(), ModelId::new("gpt-5.4") )] - .billing - .input_tokens, + .usage + .tokens + .input, 100 ); assert_eq!( @@ -16458,8 +16416,9 @@ fn aggregate_billing_counts_projection_rollup_usage_visits() { ModelId::new("gpt-5.4") ) .with_speed(Some(Speed::Fast))] - .billing - .input_tokens, + .usage + .tokens + .input, 200 ); } @@ -17797,11 +17756,10 @@ async fn attach_stream_replays_agent_message_reasoning() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -18388,7 +18346,8 @@ async fn list_runs_returns_run_list_items() { assert!(run_json_status(item).is_object()); assert!(item["timestamps"]["created_at"].is_string()); assert!(run_json_pending_control(item).is_null()); - assert!(item["billing"].is_null()); + assert_eq!(item["usage"]["tokens"]["input"], 0); + assert!(item["usage"].get("cost").is_none()); } #[tokio::test] @@ -18456,11 +18415,10 @@ async fn list_runs_excludes_archived_by_default() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, workflow_event::Event::RunArchived { actor: None }, ]) @@ -18500,11 +18458,10 @@ async fn list_runs_includes_archived_when_flag_set() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, workflow_event::Event::RunArchived { actor: None }, ]) @@ -18520,11 +18477,10 @@ async fn list_runs_includes_archived_when_flag_set() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -18578,11 +18534,10 @@ async fn get_run_exposes_canonical_operator_statuses() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -18663,11 +18618,10 @@ async fn list_runs_preserves_underlying_run_status_payloads() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -18948,11 +18902,10 @@ async fn list_runs_status_filter_accepts_repeated_values() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; @@ -19085,11 +19038,10 @@ async fn list_runs_sort_by_status_groups_by_bucket() { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }, ]) .await; diff --git a/lib/apps/fabro-server/tests/it/api/run_files.rs b/lib/apps/fabro-server/tests/it/api/run_files.rs index eb09fcf94..07514ef88 100644 --- a/lib/apps/fabro-server/tests/it/api/run_files.rs +++ b/lib/apps/fabro-server/tests/it/api/run_files.rs @@ -114,11 +114,10 @@ async fn append_completed_run_with_final_patch( artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: Some("bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb".to_string()), final_patch: Some(final_patch.to_string()), diff_summary: None, - billing: None, + usage: None, }, ) .await diff --git a/lib/apps/fabro-server/tests/it/scenario/usage.rs b/lib/apps/fabro-server/tests/it/scenario/usage.rs index 9ead50a03..a55277daa 100644 --- a/lib/apps/fabro-server/tests/it/scenario/usage.rs +++ b/lib/apps/fabro-server/tests/it/scenario/usage.rs @@ -26,7 +26,7 @@ const WAIT_DOT: &str = r#"digraph Test { }"#; #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn aggregate_billing_increments_after_run_completes() { +async fn aggregate_usage_increments_after_run_completes() { let workspace = tempfile::tempdir().unwrap(); let state = test_app_state_with_options(test_settings(), 5); let app = test_app_with_scheduler(state); @@ -45,7 +45,7 @@ async fn aggregate_billing_increments_after_run_completes() { for _ in 0..POLL_ATTEMPTS { let req = Request::builder() .method("GET") - .uri(api("/billing")) + .uri(api("/usage")) .body(Body::empty()) .unwrap(); @@ -66,7 +66,7 @@ async fn aggregate_billing_increments_after_run_completes() { } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn run_billing_includes_completed_non_llm_stages() { +async fn run_usage_includes_completed_non_llm_stages() { let workspace = tempfile::tempdir().unwrap(); let state = test_app_state_with_options(test_settings(), 5); let app = test_app_with_scheduler(state); @@ -80,12 +80,12 @@ async fn run_billing_includes_completed_non_llm_stages() { let status = wait_for_run_status(&app, &run_id, &["succeeded", "failed"]).await; assert_eq!(status, "succeeded"); - let billing = run_billing(&app, &run_id).await; - assert_non_llm_billing(&billing, &["wait_task"]); + let usage = run_usage(&app, &run_id).await; + assert_non_llm_usage(&usage, &["wait_task"]); } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn run_billing_includes_completed_command_stages() { +async fn run_usage_includes_completed_command_stages() { let workspace = tempfile::tempdir().unwrap(); let state = test_app_state_with_options(test_settings(), 5); let app = test_app_with_scheduler(state); @@ -99,30 +99,30 @@ async fn run_billing_includes_completed_command_stages() { let status = wait_for_run_status(&app, &run_id, &["succeeded", "failed"]).await; assert_eq!(status, "succeeded"); - let billing = run_billing(&app, &run_id).await; - assert_non_llm_billing(&billing, &["echo_task"]); + let usage = run_usage(&app, &run_id).await; + assert_non_llm_usage(&usage, &["echo_task"]); } -async fn run_billing(app: &axum::Router, run_id: &str) -> serde_json::Value { +async fn run_usage(app: &axum::Router, run_id: &str) -> serde_json::Value { let req = Request::builder() .method("GET") - .uri(api(&format!("/runs/{run_id}/billing"))) + .uri(api(&format!("/runs/{run_id}/usage"))) .body(Body::empty()) - .expect("run billing request should build"); + .expect("run usage request should build"); let response = app.clone().oneshot(req).await.unwrap(); crate::helpers::response_json( response, StatusCode::OK, - format!("GET /api/v1/runs/{run_id}/billing"), + format!("GET /api/v1/runs/{run_id}/usage"), ) .await } -fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&str]) { - let stages = billing["stages"] +fn assert_non_llm_usage(usage: &serde_json::Value, expected_stage_ids: &[&str]) { + let stages = usage["stages"] .as_array() - .expect("billing response should include stages"); + .expect("usage response should include stages"); let mut stage_ids = stages .iter() .map(|stage| { @@ -138,10 +138,10 @@ fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&st assert!( stages.iter().all(|stage| { stage["model"].is_null() - && stage["billing"]["input_tokens"] == 0 - && stage["billing"]["output_tokens"] == 0 - && stage["billing"]["reasoning_tokens"] == 0 - && stage["billing"]["total_usd_micros"].is_null() + && stage["usage"]["tokens"]["input"] == 0 + && stage["usage"]["tokens"]["output"] == 0 + && stage["usage"]["tokens"]["reasoning"] == 0 + && stage["usage"].get("cost").is_none() }), "every non-LLM stage should have null model and zero token counts: {stages:?}" ); @@ -156,17 +156,17 @@ fn assert_non_llm_billing(billing: &serde_json::Value, expected_stage_ids: &[&st .sum(); assert_eq!( - billing["by_model"] + usage["by_model"] .as_array() - .expect("billing response should include by_model") + .expect("usage response should include by_model") .len(), 0 ); - assert_eq!(billing["totals"]["input_tokens"], 0); - assert_eq!(billing["totals"]["output_tokens"], 0); - assert!(billing["totals"]["total_usd_micros"].is_null()); + assert_eq!(usage["totals"]["usage"]["tokens"]["input"], 0); + assert_eq!(usage["totals"]["usage"]["tokens"]["output"], 0); + assert!(usage["totals"]["usage"].get("cost").is_none()); - let total_wall_time_ms = billing["totals"]["timing"]["wall_time_ms"] + let total_wall_time_ms = usage["totals"]["timing"]["wall_time_ms"] .as_u64() .expect("totals should include timing.wall_time_ms"); assert_eq!( diff --git a/lib/components/fabro-dump/src/lib.rs b/lib/components/fabro-dump/src/lib.rs index a1395e67f..2b3e2e50b 100644 --- a/lib/components/fabro-dump/src/lib.rs +++ b/lib/components/fabro-dump/src/lib.rs @@ -558,7 +558,7 @@ mod tests { failure: None, final_git_commit_sha: Some("abc123".to_string()), stages: Vec::new(), - billing: None, + usage: None, total_retries: 0, diff: RunDiff::default(), }); diff --git a/lib/components/fabro-store/src/run_state.rs b/lib/components/fabro-store/src/run_state.rs index c5381f977..6f3868109 100644 --- a/lib/components/fabro-store/src/run_state.rs +++ b/lib/components/fabro-store/src/run_state.rs @@ -9,20 +9,19 @@ use fabro_types::run_event::{ }; use fabro_types::settings::run::RunEnvironmentSettings; use fabro_types::{ - AskFabro, BilledModelUsage, BilledTokenCounts, Checkpoint, CheckpointRecord, - CommandTermination, Conclusion, EventBody, FailureCategory, FailureSignature, - InterviewQuestionRecord, ModelRef, Outcome, PendingInterviewRecord, PendingReason, - PullRequestCreation, PullRequestCreationStatus, PullRequestLink, RepositoryRef, Run, - RunApproval, RunApprovalState, RunBillingSummary, RunControlAction, RunDiff, RunEvent, RunId, - RunLifecycle, RunLinks, RunModel, RunOrigin, RunProjection, RunSandbox, RunSandboxFailure, - RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, RunSize, RunSpec, RunStatus, - RunTimestamps, SandboxProviderKind, StageCompletion, StageHandler, StageId, + AskFabro, Checkpoint, CheckpointRecord, CommandTermination, Conclusion, EventBody, + FailureCategory, FailureSignature, InterviewQuestionRecord, ModelRef, ModelUsage, Outcome, + PendingInterviewRecord, PendingReason, PullRequestCreation, PullRequestCreationStatus, + PullRequestLink, RepositoryRef, Run, RunApproval, RunApprovalState, RunControlAction, RunDiff, + RunEvent, RunId, RunLifecycle, RunLinks, RunModel, RunOrigin, RunProjection, RunSandbox, + RunSandboxFailure, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, RunSize, RunSpec, + RunStatus, RunTimestamps, SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection, StageModelUsage, StageOutcome, StageProjection, StageState, - StartRecord, WorkflowRef, billing_rollup, first_event_seq, timing, + StartRecord, WorkflowRef, first_event_seq, sum_usage, timing, usage_rollup, }; use fabro_util::error::render_compact_with_causes; use lithos_llm::catalog::{ModelId, ProviderId}; -use lithos_llm::types::TokenCounts; +use lithos_llm::types::Usage; use pebble_coding_agent::events::CodingEvent; use pebble_coding_agent::projection::SessionProjection; @@ -501,9 +500,9 @@ impl RunProjectionReducer for RunProjection { return Ok(()); }; stage.response = Some(props.response.clone()); - if let Some(billing) = &props.billing { - stage.usage.replace_with_billed_usage(billing); - stage.model = Some(billing.model().clone()); + if let Some(usage) = &props.usage { + stage.usage = usage.usage; + stage.model = Some(usage.model().clone()); } } EventBody::StageCompleted(props) => { @@ -518,11 +517,11 @@ impl RunProjectionReducer for RunProjection { stage.response = response; stage.completion = Some(completion); stage.set_authoritative_timing(props.timing); - if let Some(billing) = &props.billing { - stage.usage.replace_with_billed_usage(billing); - stage.model = Some(billing.model().clone()); + if let Some(usage) = &props.usage { + stage.usage = usage.usage; + stage.model = Some(usage.model().clone()); } - stage.billing_by_model.clone_from(&props.billing_by_model); + stage.usage_by_model.clone_from(&props.usage_by_model); stage.state = StageState::from(outcome.status); } EventBody::StageFailed(props) => { @@ -541,11 +540,11 @@ impl RunProjectionReducer for RunProjection { timestamp: ts, }); stage.set_authoritative_timing(props.timing); - if let Some(billing) = &props.billing { - stage.usage.replace_with_billed_usage(billing); - stage.model = Some(billing.model().clone()); + if let Some(usage) = &props.usage { + stage.usage = usage.usage; + stage.model = Some(usage.model().clone()); } - stage.billing_by_model.clone_from(&props.billing_by_model); + stage.usage_by_model.clone_from(&props.usage_by_model); stage.state = stage_state_from_failure(props.will_retry, failure_category, stage.termination); } @@ -718,7 +717,7 @@ fn apply_agent_event( // fabro-only arms below read the same event. While the stage runs, its // usage is that fold's: the tree's tokens, the root's and every // subagent's, with whatever cost the provider reported. The terminal - // billing then brings the catalog's price for the same tokens. + // usage then brings the catalog's price for the same tokens. if let Some(stage) = stage_at_stored_or_visit(state, stored, visit, seq) { let agent = stage.agent.get_or_insert_default(); agent.apply(&props.event); @@ -814,17 +813,10 @@ fn apply_agent_event( } /// A running stage's usage, from its agent's fold: the tree's tokens and the -/// cost the provider reported for them, `None` when it reported none. -fn live_usage(agent: &SessionProjection) -> BilledTokenCounts { - let (descendants, descendant_cost) = agent.descendant_usage(); - let mut cost = agent.cost_usd_micros; - if let Some(descendant_cost) = descendant_cost { - cost = Some(cost.unwrap_or(0).saturating_add(descendant_cost)); - } - BilledTokenCounts::from_token_counts( - TokenCounts::from(agent.usage.saturating_add(descendants)), - cost.map(|cost| i64::try_from(cost).unwrap_or(i64::MAX)), - ) +/// cost the provider reported for them, `None` once any of them went +/// unpriced. +fn live_usage(agent: &SessionProjection) -> Usage { + agent.usage.saturating_add(agent.descendant_usage()) } /// The model reference for a message the stage's session produced. @@ -1188,7 +1180,7 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run { .conclusion .as_ref() .map(|conclusion| conclusion.timing); - let total_usd_micros = projected_billing(state).total_usd_micros; + let usage = projected_usage(state); Run { id: *run_id, @@ -1232,10 +1224,8 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run { completed_at, }, timing: run_timing, - billing: total_usd_micros.map(|total_usd_micros| RunBillingSummary { - total_usd_micros: Some(total_usd_micros), - }), - size: RunSize::from_total_usd_micros(total_usd_micros), + usage, + size: RunSize::from_cost(usage.cost), ask_fabro: AskFabro::default(), diff: diff_summary, pull_request: state.pull_request.clone(), @@ -1248,22 +1238,23 @@ pub(crate) fn build_summary(state: &RunProjection, run_id: &RunId) -> Run { } } -pub(crate) fn projected_billing(state: &RunProjection) -> BilledTokenCounts { - if let Some(billing) = state +/// The run's usage: the conclusion's total once the run ended, else the sum +/// of every non-boundary stage's usage so far. +pub(crate) fn projected_usage(state: &RunProjection) -> Usage { + if let Some(usage) = state .conclusion .as_ref() - .and_then(|conclusion| conclusion.billing.as_ref()) + .and_then(|conclusion| conclusion.usage) { - return billing.clone(); + return usage; } - let mut billing = BilledTokenCounts::default(); - for (stage_id, stage) in state.iter_stages() { - if !state.is_boundary_stage(stage_id.node_id()) { - billing.add_counts(&stage.usage); - } - } - billing + sum_usage( + state + .iter_stages() + .filter(|(stage_id, _)| !state.is_boundary_stage(stage_id.node_id())) + .map(|(_, stage)| stage.usage), + ) } fn run_models(state: &RunProjection) -> Vec { @@ -1326,7 +1317,7 @@ fn conclusion_from_completed( timestamp: DateTime, ) -> Result { let (stages, total_retries) = - billing_rollup::billing_rollup_from_projection(projection).conclusion_stages(projection); + usage_rollup::usage_rollup_from_projection(projection).conclusion_stages(projection); Ok(Conclusion { timestamp, status: StageOutcome::from_str(&props.status) @@ -1335,7 +1326,7 @@ fn conclusion_from_completed( failure: None, final_git_commit_sha: props.final_git_commit_sha.clone(), stages, - billing: props.billing.clone(), + usage: props.usage, total_retries, diff: RunDiff { patch: props.final_patch.clone(), @@ -1350,7 +1341,7 @@ fn conclusion_from_failed( timestamp: DateTime, ) -> Conclusion { let (stages, total_retries) = - billing_rollup::billing_rollup_from_projection(projection).conclusion_stages(projection); + usage_rollup::usage_rollup_from_projection(projection).conclusion_stages(projection); Conclusion { timestamp, status: StageOutcome::Failed { @@ -1360,7 +1351,7 @@ fn conclusion_from_failed( failure: Some(props.failure.clone()), final_git_commit_sha: props.final_git_commit_sha.clone(), stages, - billing: props.billing.clone(), + usage: props.usage, total_retries, diff: RunDiff { patch: props.final_patch.clone(), @@ -1432,7 +1423,7 @@ fn stage_visit( .or_else(|| state.current_visit_for(node_id)) } -fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome> { +fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome> { Outcome { status: props.status, preferred_label: props.preferred_label.clone(), @@ -1446,15 +1437,15 @@ fn stage_outcome_from_props(props: &StageCompletedProps) -> Outcome>, + outcome: &Outcome>, timestamp: DateTime, ) -> StageCompletion { StageCompletion { @@ -1511,17 +1502,17 @@ mod tests { }; use fabro_types::settings::run::DockerfileSource; use fabro_types::{ - AgentBackend, AttrValue, AutomationRef, BilledModelUsage, BilledTokenCounts, BlobHash, - BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody, - FailureCategory, FailureDetail, FailureReason, Graph, Node, Outcome, ParallelBranchId, - PendingReason, PullRequestCreationStatus, PullRequestLink, QuestionType, RunApprovalState, - RunBillingSummary, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec, RunStatus, - SandboxProviderKind, StageHandler, StageModelUsage, StageOutcome, StageState, StageTiming, - SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support, + AgentBackend, AttrValue, AutomationRef, BlobHash, BlockedReason, Checkpoint, + CheckpointRecord, CommandTermination, EventBody, FailureCategory, FailureDetail, + FailureReason, Graph, ModelUsage, Node, Outcome, ParallelBranchId, PendingReason, + PullRequestCreationStatus, PullRequestLink, QuestionType, RunApprovalState, + RunControlAction, RunDiff, RunEvent, RunSize, RunSpec, RunStatus, SandboxProviderKind, + StageHandler, StageModelUsage, StageOutcome, StageState, StageTiming, SuccessReason, + WorkflowSettings, first_event_seq, fixtures, test_support, }; - use lithos_llm::types::{ReasoningEffort, Speed, TokenCounts}; + use lithos_llm::types::{Cost, CostSource, ReasoningEffort, Speed, TokenCounts, Usage}; use pebble_coding_agent::events::{ - CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind, TokenUsage, + CodingAgentEvent, CodingEvent, CompactionReason, ErrorData, ErrorKind, }; use pebble_coding_agent::tools::ToolOutputMetadata; use serde_json::json; @@ -2060,26 +2051,24 @@ mod tests { event } - fn test_usage(model_id: &str, input_tokens: i64, output_tokens: i64) -> BilledModelUsage { + fn test_usage(model_id: &str, input_tokens: u64, output_tokens: u64) -> ModelUsage { serde_json::from_value(json!({ "model": { "provider": "openai", "model_id": model_id }, - "tokens": { - "input": input_tokens, - "output": output_tokens - }, - "total_usd_micros": input_tokens + output_tokens + "usage": { + "tokens": { + "input": input_tokens, + "output": output_tokens + }, + "cost": { "usd_micros": input_tokens + output_tokens, "source": "catalog" } + } })) .unwrap() } - fn usage_json(usage: &BilledModelUsage) -> serde_json::Value { + fn usage_json(usage: &ModelUsage) -> serde_json::Value { serde_json::to_value(usage).unwrap() } - fn usage_counts(usage: &BilledModelUsage) -> BilledTokenCounts { - BilledTokenCounts::from_billed_usage(std::slice::from_ref(usage)) - } - fn test_run_spec() -> RunSpec { RunSpec { graph_source: Some("digraph test {}".to_string()), @@ -2627,11 +2616,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, ); @@ -3318,8 +3306,8 @@ mod tests { status: StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: Some(usage.clone()), + usage_by_model: Vec::new(), + usage: Some(usage.clone()), failure: None, notes: None, files_touched: Vec::new(), @@ -3339,7 +3327,7 @@ mod tests { let stage = state.stage(&StageId::new("build", 1)).unwrap(); assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(789)); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); } @@ -3375,7 +3363,7 @@ mod tests { }, "will_retry": false, "timing": {"wall_time_ms": 654, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, - "billing": usage_json(&usage) + "usage": usage_json(&usage) }), Some("build"), )) @@ -3383,7 +3371,7 @@ mod tests { let stage = state.stage(&stage_id).unwrap(); assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(654)); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); } @@ -3406,8 +3394,8 @@ mod tests { status: StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: Some(usage), + usage_by_model: Vec::new(), + usage: Some(usage), failure: None, notes: None, files_touched: Vec::new(), @@ -3429,10 +3417,10 @@ mod tests { let first_stage = state.stage(&StageId::new("build", 1)).unwrap(); let second_stage = state.stage(&StageId::new("build", 2)).unwrap(); assert_eq!(first_stage.timing.map(|t| t.wall_time_ms), Some(111)); - assert_eq!(first_stage.usage, usage_counts(&first_usage)); + assert_eq!(first_stage.usage, first_usage.usage); assert_eq!(first_stage.model.as_ref(), Some(first_usage.model())); assert_eq!(second_stage.timing.map(|t| t.wall_time_ms), Some(222)); - assert_eq!(second_stage.usage, usage_counts(&second_usage)); + assert_eq!(second_stage.usage, second_usage.usage); assert_eq!(second_stage.model.as_ref(), Some(second_usage.model())); } @@ -3451,8 +3439,8 @@ mod tests { status: StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: Some(usage.clone()), + usage_by_model: Vec::new(), + usage: Some(usage.clone()), failure: None, notes: None, files_touched: Vec::new(), @@ -3476,7 +3464,7 @@ mod tests { ); let stage = state.stage(&scoped_stage_id).unwrap(); assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(333)); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); assert_eq!(stage.response.as_deref(), Some("done")); } @@ -3491,15 +3479,15 @@ mod tests { .apply_event(&test_stage_event( 3, EventBody::StageFailed(StageFailedProps { - index: 0, - failure: Some(fabro_types::FailureDetail::new( + index: 0, + failure: Some(fabro_types::FailureDetail::new( "try again", fabro_types::FailureCategory::TransientInfra, )), - will_retry: true, - timing: fabro_types::StageTiming::wall_only(444), - billing_by_model: Vec::new(), - billing: Some(usage.clone()), + will_retry: true, + timing: fabro_types::StageTiming::wall_only(444), + usage_by_model: Vec::new(), + usage: Some(usage.clone()), }), scoped_stage_id.clone(), )) @@ -3511,7 +3499,7 @@ mod tests { ); let stage = state.stage(&scoped_stage_id).unwrap(); assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(444)); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); let completion = stage.completion.as_ref().unwrap(); assert_eq!(completion.outcome, StageOutcome::Failed { @@ -4251,39 +4239,44 @@ mod tests { (7, "zebra", 2, 800, 200), ] { let mut props = completed_props(millis, StageOutcome::Succeeded); - props.billing = Some(test_usage("test-model", tokens, 10)); + props.usage = Some(test_usage("test-model", tokens, 10)); events.push(test_stage_event( seq, EventBody::StageCompleted(props), StageId::new(node, visit), )); } - events.push(test_raw_event(8, "checkpoint.completed", &json!({ - "status": "succeeded", - "current_node": "zebra", - "completed_nodes": ["apple", "zebra", "zebra"], - "node_retries": { "zebra": 3, "apple": 1 }, - "node_outcomes": { - "apple": Outcome::>::success(), - "zebra": Outcome::>::success(), - "skipped": Outcome::>::skipped("condition was false") - }, - "context_values": {}, - "node_visits": { "zebra": 2, "apple": 1, "skipped": 1 }, - "git_commit_sha": "checkpoint-sha" - }), Some("zebra"))); - let terminal_billing = usage_counts(&test_usage("test-model", 320, 30)); + events.push(test_raw_event( + 8, + "checkpoint.completed", + &json!({ + "status": "succeeded", + "current_node": "zebra", + "completed_nodes": ["apple", "zebra", "zebra"], + "node_retries": { "zebra": 3, "apple": 1 }, + "node_outcomes": { + "apple": Outcome::>::success(), + "zebra": Outcome::>::success(), + "skipped": Outcome::>::skipped("condition was false") + }, + "context_values": {}, + "node_visits": { "zebra": 2, "apple": 1, "skipped": 1 }, + "git_commit_sha": "checkpoint-sha" + }), + Some("zebra"), + )); + let terminal_usage = test_usage("test-model", 320, 30).usage; let terminal_props = if terminal_name == "run.completed" { json!({ "status": "succeeded", "reason": "completed", "timing": fabro_types::RunTiming::wall_only(9000), - "artifact_count": 0, "billing": terminal_billing, + "artifact_count": 0, "usage": terminal_usage, "final_git_commit_sha": "final-sha", "final_patch": "final patch" }) } else { let mut props = run_failed_props(FailureReason::WorkflowError); props.timing = fabro_types::RunTiming::wall_only(9000); - props.billing = Some(terminal_billing.clone()); + props.usage = Some(terminal_usage); props.final_git_commit_sha = Some("final-sha".to_string()); props.final_patch = Some("final patch".to_string()); serde_json::to_value(props).unwrap() @@ -4309,7 +4302,7 @@ mod tests { ); assert_eq!(conclusion.timestamp, events.last().unwrap().event.ts); assert_eq!(conclusion.timing.wall_time_ms, 9000); - assert_eq!(conclusion.billing, Some(terminal_billing)); + assert_eq!(conclusion.usage, Some(terminal_usage)); assert_eq!( conclusion.final_git_commit_sha.as_deref(), Some("final-sha") @@ -4332,7 +4325,19 @@ mod tests { "tool_time_ms": 0, "active_time_ms": 0 }, - "billing_usd_micros": 320, + "usage": { + "tokens": { + "input": 300, + "output": 20, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { + "usd_micros": 320, + "source": "catalog" + } + }, "retries": 2 }, { @@ -4344,7 +4349,19 @@ mod tests { "tool_time_ms": 0, "active_time_ms": 0 }, - "billing_usd_micros": 30, + "usage": { + "tokens": { + "input": 20, + "output": 10, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { + "usd_micros": 30, + "source": "catalog" + } + }, "retries": 0 }, { @@ -4356,6 +4373,15 @@ mod tests { "tool_time_ms": 0, "active_time_ms": 0 }, + "usage": { + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } + }, "retries": 0 } ], @@ -4397,7 +4423,7 @@ mod tests { final_git_commit_sha: Some("abc123".to_string()), final_patch: Some(patch.to_string()), diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -4543,7 +4569,7 @@ mod tests { final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -4584,7 +4610,7 @@ mod tests { final_git_commit_sha: Some("abc123".to_string()), final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -4677,11 +4703,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -4933,11 +4958,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::PartialSuccess, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -5070,11 +5094,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), None, )) @@ -5112,8 +5135,8 @@ mod tests { failure: Some(FailureDetail::new("boom", FailureCategory::TransientInfra)), will_retry, timing: fabro_types::StageTiming::wall_only(duration_ms), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, } } @@ -5123,8 +5146,8 @@ mod tests { failure: Some(FailureDetail::new("cancelled", FailureCategory::Canceled)), will_retry, timing: fabro_types::StageTiming::wall_only(duration_ms), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, } } @@ -5144,7 +5167,7 @@ mod tests { final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, } } @@ -5164,8 +5187,8 @@ mod tests { status, preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -5181,19 +5204,21 @@ mod tests { } } - fn billed_usage() -> BilledModelUsage { + fn priced_usage() -> ModelUsage { serde_json::from_value(json!({ "model": { "provider": "openai", "model_id": "gpt-test" }, - "tokens": { - "input": 10, - "output": 5, - "reasoning": 2, - "cache_read": 3, - "cache_write": 4 - }, - "total_usd_micros": 123 + "usage": { + "tokens": { + "input": 10, + "output": 5, + "reasoning": 2, + "cache_read": 3, + "cache_write": 4 + }, + "cost": { "usd_micros": 123, "source": "catalog" } + } })) - .expect("billing fixture should deserialize") + .expect("usage fixture should deserialize") } fn agent_body(event: CodingEvent) -> EventBody { @@ -5207,14 +5232,12 @@ mod tests { fn assistant_message(input: u64, output: u64) -> CodingEvent { CodingEvent::AssistantMessage { text: "assistant text".to_string(), - model: billed_usage().model().model_id.to_string(), - usage: TokenUsage { + model: priced_usage().model().model_id.to_string(), + usage: Usage::from(TokenCounts { input, output, - ..TokenUsage::default() - }, - cost_usd_micros: None, - cost_source: None, + ..TokenCounts::default() + }), tool_call_count: 0, context_window: None, reasoning: None, @@ -5238,16 +5261,12 @@ mod tests { }) } - fn live_counts(input_tokens: i64, output_tokens: i64) -> BilledTokenCounts { - BilledTokenCounts { - input_tokens, - output_tokens, - total_tokens: input_tokens + output_tokens, - reasoning_tokens: 0, - cache_read_tokens: 0, - cache_write_tokens: 0, - total_usd_micros: None, - } + fn live_counts(input: u64, output: u64) -> Usage { + Usage::from(TokenCounts { + input, + output, + ..TokenCounts::default() + }) } #[test] @@ -5273,7 +5292,7 @@ mod tests { fn agent_message_accumulates_live_usage_on_stage_projection() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); - let model = billed_usage().model().clone(); + let model = priced_usage().model().clone(); state .apply_event(&test_stage_event( @@ -5323,14 +5342,14 @@ mod tests { } /// One usage rule: a stage's usage is its session tree's, live and at - /// completion. The terminal billing carries the tokens the fold already + /// completion. The terminal usage carries the tokens the fold already /// showed plus the catalog's price, so completion changes the cost, not /// the tokens, and keeps the split by model. #[test] fn stage_completed_keeps_the_trees_live_usage_and_prices_it() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); - let model = billed_usage().model().clone(); + let model = priced_usage().model().clone(); state .apply_event(&test_stage_event( @@ -5360,25 +5379,27 @@ mod tests { stage_id.clone(), )) .unwrap(); - let live = state.stage(&stage_id).unwrap().usage.clone(); + let live = state.stage(&stage_id).unwrap().usage; assert_eq!( live, live_counts(107, 51), "the subagent's tokens are the stage's too" ); - let tree = BilledModelUsage { - model: model.clone(), - tokens: TokenCounts { + let tree = ModelUsage::new(model.clone(), Usage { + tokens: TokenCounts { input: 107, output: 51, ..TokenCounts::default() }, - total_usd_micros: Some(321), - }; + cost: Some(Cost { + usd_micros: 321, + source: CostSource::Catalog, + }), + }); let mut props = completed_props(42, StageOutcome::Succeeded); - props.billing = Some(tree.clone()); - props.billing_by_model = vec![tree.clone()]; + props.usage = Some(tree.clone()); + props.usage_by_model = vec![tree.clone()]; state .apply_event(&test_stage_event( 5, @@ -5389,17 +5410,19 @@ mod tests { let stage = state.stage(&stage_id).unwrap(); assert_eq!( - stage.usage.token_counts(), - live.token_counts(), + stage.usage.tokens, live.tokens, "completion keeps the tokens the fold showed" ); assert_eq!( - stage.usage.total_usd_micros, - Some(321), + stage.usage.cost, + Some(Cost { + usd_micros: 321, + source: CostSource::Catalog, + }), "and brings the catalog's price" ); assert_eq!(stage.model.as_ref(), Some(&model)); - assert_eq!(stage.billing_by_model, vec![tree]); + assert_eq!(stage.usage_by_model, vec![tree]); } #[test] @@ -5411,7 +5434,6 @@ mod tests { text, model, usage, - cost_source, tool_call_count, context_window, reasoning, @@ -5423,9 +5445,13 @@ mod tests { agent_body(CodingEvent::AssistantMessage { text, model, - usage, - cost_usd_micros: Some(cost), - cost_source, + usage: Usage { + tokens: usage.tokens, + cost: Some(Cost { + usd_micros: cost, + source: CostSource::Provider, + }), + }, tool_call_count, context_window, reasoning, @@ -5462,11 +5488,16 @@ mod tests { summary_token_estimate: 500, tracked_file_count: 1, reason: CompactionReason::Threshold, - usage: TokenUsage { - input: 30, - ..TokenUsage::default() + usage: Usage { + tokens: TokenCounts { + input: 30, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: 2, + source: CostSource::Provider, + }), }, - cost_usd_micros: Some(2), }), stage_id.clone(), )) @@ -5474,20 +5505,21 @@ mod tests { let stage = state.stage(&stage_id).unwrap(); assert_eq!( - stage.usage, - BilledTokenCounts { - total_usd_micros: Some(7), - ..live_counts(47, 6) - }, - "the root's messages and compaction, the child's message, and the provider's cost" + stage.usage.tokens, + live_counts(47, 6).tokens, + "the root's messages and compaction, and the child's message" + ); + assert_eq!( + stage.usage.cost, None, + "the child's unpriced message leaves the tree's cost unknown" ); } #[test] - fn stage_completed_without_billing_preserves_live_usage() { + fn stage_completed_without_usage_preserves_live_usage() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); - let model = billed_usage().model().clone(); + let model = priced_usage().model().clone(); state .apply_event(&test_stage_event( @@ -5537,7 +5569,7 @@ mod tests { )) .unwrap(); let mut props = completed_props(42, StageOutcome::Succeeded); - props.billing = Some(usage); + props.usage = Some(usage); state .apply_event(&test_stage_event( 2, @@ -5549,18 +5581,16 @@ mod tests { let summary = build_summary(&state, &fixtures::RUN_1); assert_eq!(summary.size, RunSize::S); assert_eq!( - summary.billing, - Some(RunBillingSummary { - total_usd_micros: Some(20_000_001), - }) + summary.usage.cost.map(|cost| cost.usd_micros), + Some(20_000_001) ); } #[test] - fn stage_failed_replaces_live_usage_with_terminal_billing() { + fn stage_failed_replaces_live_usage_with_terminal_usage() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); - let usage = billed_usage(); + let usage = priced_usage(); state .apply_event(&test_stage_event( @@ -5577,7 +5607,7 @@ mod tests { )) .unwrap(); let mut props = failed_props(42, false); - props.billing = Some(usage.clone()); + props.usage = Some(usage.clone()); state .apply_event(&test_stage_event( 3, @@ -5587,7 +5617,7 @@ mod tests { .unwrap(); let stage = state.stage(&stage_id).unwrap(); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); } @@ -5619,7 +5649,7 @@ mod tests { .unwrap(); let stage = state.stage(&stage_id).unwrap(); - assert!(stage.usage.is_zero()); + assert_eq!(stage.usage, Usage::default()); assert_eq!(stage.model, None); assert_eq!(stage.state, StageState::Running); } @@ -5628,7 +5658,7 @@ mod tests { fn stage_completed_records_duration_usage_and_terminal_state() { let mut state = initialized_projection(); let stage_id = StageId::new("build", 1); - let usage = billed_usage(); + let usage = priced_usage(); state .apply_event(&test_stage_event( @@ -5638,7 +5668,7 @@ mod tests { )) .unwrap(); let mut props = completed_props(42, StageOutcome::Succeeded); - props.billing = Some(usage.clone()); + props.usage = Some(usage.clone()); state .apply_event(&test_event( 2, @@ -5649,7 +5679,7 @@ mod tests { let stage = state.stage(&stage_id).unwrap(); assert_eq!(stage.timing.map(|t| t.wall_time_ms), Some(42)); - assert_eq!(stage.usage, usage_counts(&usage)); + assert_eq!(stage.usage, usage.usage); assert_eq!(stage.model.as_ref(), Some(usage.model())); assert_eq!(stage.state, StageState::Succeeded); assert_eq!(stage.effective_state(), StageState::Succeeded); @@ -5735,15 +5765,15 @@ mod tests { .apply_event(&test_event( 3, EventBody::StageFailed(StageFailedProps { - index: 0, - failure: Some(FailureDetail::new( + index: 0, + failure: Some(FailureDetail::new( "Script failed with exit code: 100\n\nCancelling due to test failure", FailureCategory::Canceled, )), - will_retry: false, - timing: fabro_types::StageTiming::wall_only(10), - billing_by_model: Vec::new(), - billing: None, + will_retry: false, + timing: fabro_types::StageTiming::wall_only(10), + usage_by_model: Vec::new(), + usage: None, }), Some("build"), )) @@ -6435,7 +6465,7 @@ mod tests { assert!(open_bracket(&state).is_none()); // The close must not undo the rest of the message's work. - assert_eq!(state.stage(&stage_id()).unwrap().usage.input_tokens, 10); + assert_eq!(state.stage(&stage_id()).unwrap().usage.tokens.input, 10); } #[test] diff --git a/lib/components/fabro-store/src/run_summary_store.rs b/lib/components/fabro-store/src/run_summary_store.rs index 191a897ce..171986587 100644 --- a/lib/components/fabro-store/src/run_summary_store.rs +++ b/lib/components/fabro-store/src/run_summary_store.rs @@ -3,8 +3,8 @@ use std::sync::LazyLock; use chrono::{DateTime, Utc}; use fabro_types::{ - BilledTokenCounts, EventEnvelope, Run, RunEvent, RunId, RunSize, RunStatusKind, RunTiming, - SessionId, StageId, timing, + EventEnvelope, Run, RunEvent, RunId, RunSize, RunStatusKind, RunTiming, SessionId, StageId, + timing, }; use sqlx::pool::PoolConnection; use sqlx::query::Query; @@ -12,7 +12,7 @@ use sqlx::sqlite::{SqliteArguments, SqliteConnection, SqliteRow}; use sqlx::{Connection as _, QueryBuilder, Row as _, Sqlite, SqlitePool, Transaction}; use strum::VariantArray as _; -use crate::run_state::{ProjectedRun, build_summary, projected_billing}; +use crate::run_state::{ProjectedRun, build_summary, projected_usage}; use crate::{Error, EventPayload, Result, keys}; const INSERT_RUN_SQL: &str = r" @@ -1019,7 +1019,7 @@ impl PreparedRunSummary { .unwrap_or(run.timestamps.created_at); run.timing = entry.projection.live_run_timing(at); } - let billing = normalize_billing_for_read_model(projected_billing(&entry.projection)); + let usage = projected_usage(&entry.projection); let workflow_name = run.workflow.display_name().map(str::to_string); let repository_name = run .repository @@ -1031,12 +1031,12 @@ impl PreparedRunSummary { last_seq: entry.last_seq, workflow_name, repository_name, - input_tokens: billing.input_tokens, - output_tokens: billing.output_tokens, - reasoning_tokens: billing.reasoning_tokens, - cache_read_tokens: billing.cache_read_tokens, - cache_write_tokens: billing.cache_write_tokens, - total_usd_micros: billing.total_usd_micros, + input_tokens: column_count(usage.tokens.input), + output_tokens: column_count(usage.tokens.output), + reasoning_tokens: column_count(usage.tokens.reasoning), + cache_read_tokens: column_count(usage.tokens.cache_read), + cache_write_tokens: column_count(usage.tokens.cache_write), + total_usd_micros: usage.cost.map(|cost| column_count(cost.usd_micros)), } } } @@ -1250,31 +1250,10 @@ async fn select_run_head(connection: &mut SqliteConnection, run_id: &RunId) -> R .transpose() } -/// Older provider codecs could persist a negative disjoint bucket when a -/// detail count exceeded its inclusive parent total. The SQLite summary is a -/// rebuildable, nonnegative read model, so normalize those legacy values here -/// without rewriting the authoritative run events. -fn normalize_billing_for_read_model(mut billing: BilledTokenCounts) -> BilledTokenCounts { - let input_total = billing - .input_tokens - .saturating_add(billing.cache_read_tokens) - .saturating_add(billing.cache_write_tokens) - .max(0); - billing.cache_read_tokens = billing.cache_read_tokens.clamp(0, input_total); - billing.cache_write_tokens = billing - .cache_write_tokens - .clamp(0, input_total - billing.cache_read_tokens); - billing.input_tokens = input_total - billing.cache_read_tokens - billing.cache_write_tokens; - - let output_total = billing - .output_tokens - .saturating_add(billing.reasoning_tokens) - .max(0); - billing.reasoning_tokens = billing.reasoning_tokens.clamp(0, output_total); - billing.output_tokens = output_total - billing.reasoning_tokens; - billing.total_tokens = input_total.saturating_add(output_total); - billing.total_usd_micros = billing.total_usd_micros.map(|value| value.max(0)); - billing +/// A usage count as the SQLite read model stores it: the columns are signed, +/// so a count past `i64::MAX` saturates rather than wrapping negative. +fn column_count(count: u64) -> i64 { + i64::try_from(count).unwrap_or(i64::MAX) } #[cfg(test)] @@ -1538,11 +1517,12 @@ mod tests { use chrono::{DateTime, Utc}; use fabro_types::{ - AutomationRef, BilledTokenCounts, BlockedReason, Conclusion, DiffSummary, EventEnvelope, - FailureReason, Graph, PendingReason, PullRequestCreationId, RunDiff, RunId, RunProjection, - RunSize, RunSpec, RunStatus, RunStatusKind, RunTiming, SessionId, StageId, StageOutcome, - SuccessReason, WorkflowSettings, test_support, + AutomationRef, BlockedReason, Conclusion, DiffSummary, EventEnvelope, FailureReason, Graph, + PendingReason, PullRequestCreationId, RunDiff, RunId, RunProjection, RunSize, RunSpec, + RunStatus, RunStatusKind, RunTiming, SessionId, StageId, StageOutcome, SuccessReason, + WorkflowSettings, test_support, }; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use strum::VariantArray as _; use tokio::time; use ulid::Ulid; @@ -2908,7 +2888,7 @@ mod tests { } #[tokio::test] - async fn projection_persists_billing_diff_and_derived_size() { + async fn projection_persists_usage_diff_and_derived_size() { let (_directory, store) = store().await; let created_at = dt("2026-07-11T12:00:00Z"); let run_id = run_id(created_at.timestamp_millis().cast_unsigned(), 1); @@ -2930,14 +2910,18 @@ mod tests { failure: None, final_git_commit_sha: None, stages: Vec::new(), - billing: Some(BilledTokenCounts { - input_tokens: 100, - output_tokens: 20, - total_tokens: 135, - reasoning_tokens: 5, - cache_read_tokens: 10, - cache_write_tokens: 0, - total_usd_micros: Some(21_000_000), + usage: Some(Usage { + tokens: TokenCounts { + input: 100, + output: 20, + reasoning: 5, + cache_read: 10, + cache_write: 0, + }, + cost: Some(Cost { + usd_micros: 21_000_000, + source: CostSource::Catalog, + }), }), total_retries: 0, diff: RunDiff { @@ -2997,47 +2981,6 @@ mod tests { assert_eq!(run.size, RunSize::S); } - #[tokio::test] - async fn projection_normalizes_legacy_overlapping_reasoning_tokens() { - let (_directory, store) = store().await; - let created_at = dt("2026-07-11T12:00:00Z"); - let run_id = run_id(created_at.timestamp_millis().cast_unsigned(), 1); - let mut projection = projection(run_id, "legacy billing", created_at); - projection.conclusion = Some(Conclusion { - timestamp: created_at, - status: StageOutcome::Succeeded, - timing: RunTiming::default(), - failure: None, - final_git_commit_sha: None, - stages: Vec::new(), - billing: Some(BilledTokenCounts { - input_tokens: 53, - output_tokens: -7, - total_tokens: 112, - reasoning_tokens: 66, - ..BilledTokenCounts::default() - }), - total_retries: 0, - diff: RunDiff::default(), - }); - - store - .upsert_projection(&entry(projection, 1)) - .await - .unwrap(); - - let row = sqlx::query( - "SELECT input_tokens, output_tokens, reasoning_tokens FROM runs WHERE id = ?", - ) - .bind(run_id.to_string()) - .fetch_one(&store.pool) - .await - .unwrap(); - assert_eq!(sqlx::Row::get::(&row, "input_tokens"), 53); - assert_eq!(sqlx::Row::get::(&row, "output_tokens"), 0); - assert_eq!(sqlx::Row::get::(&row, "reasoning_tokens"), 59); - } - #[tokio::test] async fn reconcile_removes_rows_absent_from_authoritative_entries() { let (_directory, store) = store().await; diff --git a/lib/components/fabro-store/tests/serializable_projection.rs b/lib/components/fabro-store/tests/serializable_projection.rs index 5608295e7..a559681c6 100644 --- a/lib/components/fabro-store/tests/serializable_projection.rs +++ b/lib/components/fabro-store/tests/serializable_projection.rs @@ -5,10 +5,10 @@ use fabro_store::{RunProjection, SerializableProjection, StageId}; use fabro_types::graph::Graph; use fabro_types::run::RunSpec; use fabro_types::{ - BilledModelUsage, BilledTokenCounts, Checkpoint, CheckpointRecord, InterviewQuestionRecord, - ParallelBranchResult, QuestionType, RunDiff, RunSandbox, RunSandboxInstance, RunSandboxPlan, - RunSandboxRuntime, RunStatus, SandboxProviderKind, StageCompletion, StageModelUsage, - StageOutcome, StartRecord, first_event_seq, fixtures, test_support, + Checkpoint, CheckpointRecord, InterviewQuestionRecord, ModelUsage, ParallelBranchResult, + QuestionType, RunDiff, RunSandbox, RunSandboxInstance, RunSandboxPlan, RunSandboxRuntime, + RunStatus, SandboxProviderKind, StageCompletion, StageModelUsage, StageOutcome, StartRecord, + first_event_seq, fixtures, test_support, }; use serde_json::json; @@ -47,14 +47,16 @@ fn sample_checkpoint() -> Checkpoint { } } -fn sample_usage() -> BilledModelUsage { +fn sample_usage() -> ModelUsage { serde_json::from_value(json!({ "model": { "provider": "openai", "model_id": "gpt-5.2" }, - "tokens": { - "input": 123, - "output": 45 - }, - "total_usd_micros": 168 + "usage": { + "tokens": { + "input": 123, + "output": 45 + }, + "cost": { "usd_micros": 168, "source": "catalog" } + } })) .expect("sample usage should deserialize") } @@ -137,15 +139,14 @@ fn serializable_projection_round_trips_and_trims_bulky_node_fields() { stage.parallel_results = Some(parallel_results.clone()); stage.timing = Some(fabro_types::StageTiming::wall_only(1234)); let usage = sample_usage(); - let usage_counts = BilledTokenCounts::from_billed_usage(std::slice::from_ref(&usage)); - stage.usage = usage_counts.clone(); + stage.usage = usage.usage; stage.model = Some(usage.model().clone()); stage.output = Some("output".to_string()); let serialized = serde_json::to_value(SerializableProjection(&projection)) .expect("projection should serialize"); assert_eq!( - serialized["stages"]["build@2"]["usage"]["input_tokens"], + serialized["stages"]["build@2"]["usage"]["tokens"]["input"], json!(123) ); assert_eq!( @@ -196,7 +197,7 @@ fn serializable_projection_round_trips_and_trims_bulky_node_fields() { assert_eq!(node.script_timing, Some(json!({ "duration_ms": 10 }))); assert_eq!(node.parallel_results, Some(parallel_results)); assert_eq!(node.timing.map(|t| t.wall_time_ms), Some(1234)); - assert_eq!(node.usage, usage_counts); + assert_eq!(node.usage, usage.usage); assert_eq!(node.model.as_ref(), Some(usage.model())); } diff --git a/lib/components/fabro-tool/src/common.rs b/lib/components/fabro-tool/src/common.rs index 7b92b3a09..b24128ff9 100644 --- a/lib/components/fabro-tool/src/common.rs +++ b/lib/components/fabro-tool/src/common.rs @@ -311,6 +311,7 @@ fn format_tool_error(err: &anyhow::Error) -> String { #[cfg(test)] mod tests { use chrono::{TimeZone, Utc}; + use fabro_api::types::Usage; use fabro_types::{ RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef, test_support, }; @@ -457,7 +458,7 @@ mod tests { completed_at: None, }, timing: None, - billing: None, + usage: Usage::default(), size: fabro_types::RunSize::default(), ask_fabro: fabro_types::AskFabro::default(), diff: None, diff --git a/lib/components/fabro-tool/src/create.rs b/lib/components/fabro-tool/src/create.rs index 21449d610..d3b23d30b 100644 --- a/lib/components/fabro-tool/src/create.rs +++ b/lib/components/fabro-tool/src/create.rs @@ -279,6 +279,7 @@ mod tests { use std::collections::HashMap; use chrono::{TimeZone, Utc}; + use fabro_api::types::Usage; use fabro_types::{ GitRunTarget, Run, RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef, test_support, @@ -639,7 +640,7 @@ mod tests { completed_at: None, }, timing: None, - billing: None, + usage: Usage::default(), size: fabro_types::RunSize::default(), ask_fabro: fabro_types::AskFabro::default(), diff: None, diff --git a/lib/components/fabro-tool/src/interact.rs b/lib/components/fabro-tool/src/interact.rs index 0757cdb51..2a780361c 100644 --- a/lib/components/fabro-tool/src/interact.rs +++ b/lib/components/fabro-tool/src/interact.rs @@ -450,6 +450,7 @@ mod tests { use async_trait::async_trait; use chrono::{TimeZone, Utc}; + use fabro_api::types::Usage; use fabro_types::{ EventEnvelope, FailureReason, Run, RunId, RunLifecycle, RunLinks, RunOrigin, RunProjection, RunStatus, RunTimestamps, WorkflowRef, test_support, @@ -711,7 +712,7 @@ mod tests { completed_at: None, }, timing: None, - billing: None, + usage: Usage::default(), size: fabro_types::RunSize::default(), ask_fabro: fabro_types::AskFabro::default(), diff: None, diff --git a/lib/components/fabro-tool/src/search.rs b/lib/components/fabro-tool/src/search.rs index e3d0162c6..7e72a93db 100644 --- a/lib/components/fabro-tool/src/search.rs +++ b/lib/components/fabro-tool/src/search.rs @@ -293,6 +293,7 @@ mod tests { use std::collections::HashMap; use chrono::{TimeZone, Utc}; + use fabro_api::types::Usage; use fabro_types::{ RunLifecycle, RunLinks, RunOrigin, RunStatus, RunTimestamps, WorkflowRef, test_support, }; @@ -468,7 +469,7 @@ mod tests { completed_at: None, }, timing: None, - billing: None, + usage: Usage::default(), size: fabro_types::RunSize::default(), ask_fabro: fabro_types::AskFabro::default(), diff: None, diff --git a/lib/components/fabro-workflow/src/error.rs b/lib/components/fabro-workflow/src/error.rs index a0e85035b..67f43a09d 100644 --- a/lib/components/fabro-workflow/src/error.rs +++ b/lib/components/fabro-workflow/src/error.rs @@ -2128,15 +2128,15 @@ mod tests { // 3. Outcome → StageFailed event let failure = outcome.failure.clone().unwrap(); let event = Event::StageFailed { - node_id: "code".into(), - name: "code".into(), - index: 0, - failure: failure.clone(), - will_retry: false, - timing: fabro_types::StageTiming::wall_only(0), - billing_by_model: Vec::new(), - billing: None, - actor: None, + node_id: "code".into(), + name: "code".into(), + index: 0, + failure: failure.clone(), + will_retry: false, + timing: fabro_types::StageTiming::wall_only(0), + usage_by_model: Vec::new(), + usage: None, + actor: None, }; // 4. Verify classification survived all the way through diff --git a/lib/components/fabro-workflow/src/event/convert.rs b/lib/components/fabro-workflow/src/event/convert.rs index e74f22813..02c803ffc 100644 --- a/lib/components/fabro-workflow/src/event/convert.rs +++ b/lib/components/fabro-workflow/src/event/convert.rs @@ -235,21 +235,19 @@ fn event_body_from_event(event: &Event) -> EventBody { artifact_count, status, reason, - total_usd_micros, final_git_commit_sha, final_patch, diff_summary, - billing, + usage, } => EventBody::RunCompleted(fabro_types::RunCompletedProps { timing: *timing, artifact_count: *artifact_count, status: status.clone(), reason: *reason, - total_usd_micros: *total_usd_micros, final_git_commit_sha: final_git_commit_sha.clone(), final_patch: final_patch.clone(), diff_summary: *diff_summary, - billing: billing.clone(), + usage: *usage, }), Event::WorkflowRunFailed { failure, @@ -257,14 +255,14 @@ fn event_body_from_event(event: &Event) -> EventBody { final_git_commit_sha, final_patch, diff_summary, - billing, + usage, } => EventBody::RunFailed(fabro_types::RunFailedProps { failure: failure.clone(), timing: *timing, final_git_commit_sha: final_git_commit_sha.clone(), final_patch: final_patch.clone(), diff_summary: *diff_summary, - billing: billing.clone(), + usage: *usage, }), Event::RunNotice { level, @@ -343,8 +341,8 @@ fn event_body_from_event(event: &Event) -> EventBody { status, preferred_label, suggested_next_ids, - billing, - billing_by_model, + usage, + usage_by_model, failure, notes, files_touched, @@ -364,8 +362,8 @@ fn event_body_from_event(event: &Event) -> EventBody { status: stage_status_from_string(status), preferred_label: preferred_label.clone(), suggested_next_ids: suggested_next_ids.clone(), - billing: billing.clone(), - billing_by_model: billing_by_model.clone(), + usage: usage.clone(), + usage_by_model: usage_by_model.clone(), failure: failure.clone(), notes: notes.clone(), files_touched: files_touched.clone(), @@ -384,16 +382,16 @@ fn event_body_from_event(event: &Event) -> EventBody { failure, will_retry, timing, - billing, - billing_by_model, + usage, + usage_by_model, .. } => EventBody::StageFailed(fabro_types::StageFailedProps { index: *index, failure: Some(failure.clone()), will_retry: *will_retry, timing: *timing, - billing: billing.clone(), - billing_by_model: billing_by_model.clone(), + usage: usage.clone(), + usage_by_model: usage_by_model.clone(), }), Event::StageRetrying { index, @@ -624,13 +622,13 @@ fn event_body_from_event(event: &Event) -> EventBody { response, model, provider, - billing, + usage, .. } => EventBody::PromptCompleted(fabro_types::PromptCompletedProps { response: response.clone(), model: model.clone(), provider: provider.clone(), - billing: billing.clone(), + usage: usage.clone(), }), Event::Agent { stage, @@ -1021,7 +1019,9 @@ mod tests { }; use chrono::Utc; use lithos_llm::types::ReasoningOutput; - use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, TokenUsage}; + use pebble_coding_agent::events::{ + CodingAgentEvent, CodingEvent, Cost, CostSource, TokenCounts, Usage, + }; use super::*; use crate::error::Error; @@ -1063,8 +1063,8 @@ mod tests { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -1109,8 +1109,8 @@ mod tests { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -1135,18 +1135,18 @@ mod tests { fn run_event_stage_failure_keeps_failure_detail() { let usage = test_usage("gpt-5.2", 321, 54); let stored = to_run_event(&fixtures::RUN_3, &Event::StageFailed { - node_id: "code".to_string(), - name: "Code".to_string(), - index: 1, - failure: FailureDetail::new( + node_id: "code".to_string(), + name: "Code".to_string(), + index: 1, + failure: FailureDetail::new( "lint failed", crate::outcome::FailureCategory::Deterministic, ), - will_retry: true, - timing: ::fabro_types::StageTiming::wall_only(5000), - billing_by_model: Vec::new(), - billing: Some(usage.clone()), - actor: None, + will_retry: true, + timing: ::fabro_types::StageTiming::wall_only(5000), + usage_by_model: Vec::new(), + usage: Some(usage.clone()), + actor: None, }); assert_eq!(stored.event_name(), "stage.failed"); @@ -1154,7 +1154,7 @@ mod tests { assert_eq!(properties["failure"]["message"], "lint failed"); assert_eq!(properties["failure"]["category"], "deterministic"); assert_eq!(properties["will_retry"], true); - assert_eq!(properties["billing"], serde_json::to_value(&usage).unwrap()); + assert_eq!(properties["usage"], serde_json::to_value(&usage).unwrap()); } #[test] @@ -2252,9 +2252,7 @@ mod tests { event: agent_event("ses_agent", CodingEvent::AssistantMessage { text: "ok".to_string(), model: "claude-sonnet".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window: None, reasoning: None, @@ -2278,9 +2276,13 @@ mod tests { event: agent_event("ses_agent", CodingEvent::AssistantMessage { text: String::new(), model: "gpt-5.4".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: Some(125_000), - cost_source: Some(pebble_coding_agent::events::CostSource::Provider), + usage: Usage { + tokens: TokenCounts::default(), + cost: Some(Cost { + usd_micros: 125_000, + source: CostSource::Provider, + }), + }, tool_call_count: 1, context_window: None, reasoning: Some(ReasoningOutput::new( @@ -2293,7 +2295,8 @@ mod tests { let value = stored.to_value().unwrap(); assert_eq!(value["event"], "agent.message"); let message = &value["properties"]["event"]["AssistantMessage"]; - assert_eq!(message["cost_usd_micros"], 125_000); + assert_eq!(message["usage"]["cost"]["usd_micros"], 125_000); + assert_eq!(message["usage"]["cost"]["source"], "provider"); assert_eq!( message["reasoning"]["summary"], "inspect the conversion first" diff --git a/lib/components/fabro-workflow/src/event/events.rs b/lib/components/fabro-workflow/src/event/events.rs index 69ea3e38e..95b9a4a29 100644 --- a/lib/components/fabro-workflow/src/event/events.rs +++ b/lib/components/fabro-workflow/src/event/events.rs @@ -1,20 +1,20 @@ use std::collections::BTreeMap; use ::fabro_types::{ - AutomationRef, BilledTokenCounts, BlobHash, BlockedReason, CommandTermination, DiffSummary, - FailureReason, ForkSourceRef, GitContext, PairId, PairMessageId, PairSystemMessageKind, - PairTarget, ParallelBranchId, ParallelBranchResult, PendingReason, PermissionLevel, Principal, + AutomationRef, BlobHash, BlockedReason, CommandTermination, DiffSummary, FailureReason, + ForkSourceRef, GitContext, PairId, PairMessageId, PairSystemMessageKind, PairTarget, + ParallelBranchId, ParallelBranchResult, PendingReason, PermissionLevel, Principal, PullRequestCreationId, PullRequestLink, ReviewTarget, RunFailure, RunId, RunNoticeLevel, RunPairEndedReason, RunPairFailedReason, RunProvenance, RunRunnableSource, RunTarget, RunTiming, SandboxProviderKind, StageId, StageOutcome, StageTiming, SuccessReason, WorkflowVersionId, run_event as fabro_types, }; -use lithos_llm::types::{ReasoningEffort, Speed}; +use lithos_llm::types::{ReasoningEffort, Speed, Usage}; use pebble_coding_agent::events::CodingAgentEvent; use serde::{Deserialize, Serialize}; use crate::error::{Error, run_failure_from_error}; -use crate::outcome::{BilledModelUsage, FailureDetail, Outcome}; +use crate::outcome::{FailureDetail, ModelUsage, Outcome}; /// Events emitted during workflow run execution for observability. #[derive(Debug, Clone, Serialize, Deserialize)] @@ -185,15 +185,13 @@ pub enum Event { status: String, reason: SuccessReason, #[serde(default, skip_serializing_if = "Option::is_none")] - total_usd_micros: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] final_git_commit_sha: Option, #[serde(default, skip_serializing_if = "Option::is_none")] final_patch: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff_summary: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - billing: Option, + usage: Option, }, WorkflowRunFailed { failure: RunFailure, @@ -205,7 +203,7 @@ pub enum Event { #[serde(default, skip_serializing_if = "Option::is_none")] diff_summary: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - billing: Option, + usage: Option, }, RunNotice { level: RunNoticeLevel, @@ -269,9 +267,9 @@ pub enum Event { status: String, preferred_label: Option, suggested_next_ids: Vec, - billing: Option, + usage: Option, #[serde(default, skip_serializing_if = "Vec::is_empty")] - billing_by_model: Vec, + usage_by_model: Vec, #[serde(default, skip_serializing_if = "Option::is_none")] failure: Option, notes: Option, @@ -294,17 +292,17 @@ pub enum Event { max_attempts: usize, }, StageFailed { - node_id: String, - name: String, - index: usize, - failure: FailureDetail, - will_retry: bool, - timing: StageTiming, - billing: Option, + node_id: String, + name: String, + index: usize, + failure: FailureDetail, + will_retry: bool, + timing: StageTiming, + usage: Option, #[serde(default, skip_serializing_if = "Vec::is_empty")] - billing_by_model: Vec, + usage_by_model: Vec, #[serde(default, skip_serializing_if = "Option::is_none")] - actor: Option, + actor: Option, }, StageRetrying { node_id: String, @@ -495,7 +493,7 @@ pub enum Event { model: String, provider: String, #[serde(default, skip_serializing_if = "Option::is_none")] - billing: Option, + usage: Option, }, /// One coding-agent event, tagged with the workflow stage that produced /// it. Pebble's envelope is kept whole: `seq`, `stream_id`, session ids, @@ -819,7 +817,7 @@ impl Event { final_git_commit_sha: Option, final_patch: Option, diff_summary: Option, - billing: Option, + usage: Option, ) -> Self { Self::WorkflowRunFailed { failure: run_failure_from_error(error, reason), @@ -827,7 +825,7 @@ impl Event { final_git_commit_sha, final_patch, diff_summary, - billing, + usage, } } diff --git a/lib/components/fabro-workflow/src/event/redaction.rs b/lib/components/fabro-workflow/src/event/redaction.rs index 37e288cd9..f5de7b104 100644 --- a/lib/components/fabro-workflow/src/event/redaction.rs +++ b/lib/components/fabro-workflow/src/event/redaction.rs @@ -32,7 +32,7 @@ pub fn event_payload_from_redacted_json(line: &str, run_id: &RunId) -> Result; + type Meta = Option; fn get_node(&self, id: &str) -> Option { self.0 diff --git a/lib/components/fabro-workflow/src/handler/agent.rs b/lib/components/fabro-workflow/src/handler/agent.rs index 0389037de..ba187d74c 100644 --- a/lib/components/fabro-workflow/src/handler/agent.rs +++ b/lib/components/fabro-workflow/src/handler/agent.rs @@ -19,7 +19,7 @@ use crate::context::{Context, WorkflowContext, keys}; use crate::error::Error; use crate::event::{Emitter, Event, StageScope}; use crate::interview_runtime::WorkflowHumanInput; -use crate::outcome::{BilledModelUsage, Outcome, OutcomeExt}; +use crate::outcome::{ModelUsage, Outcome, OutcomeExt}; const LAST_FILE_ROUTING_EXTENSIONS: &[&str] = &["json", "md"]; @@ -31,12 +31,12 @@ const LAST_FILE_ROUTING_EXTENSIONS: &[&str] = &["json", "md"]; pub enum CodergenResult { Text { text: String, - /// The stage's billing: for an agent, the whole session tree's + /// The stage's usage: for an agent, the whole session tree's /// tokens under the root's route. - usage: Option, + usage: Option, /// `usage` split by model, when the backend billed subagents at /// their own models. Empty when `usage` is the one row. - usage_by_model: Vec, + usage_by_model: Vec, files_touched: Vec, last_file_touched: Option, /// Active timing observed by the backend. The wall field is ignored by @@ -380,7 +380,7 @@ impl Handler for AgentHandler { response: response_text.clone(), model: response_model, provider: response_provider, - billing: stage_usage.clone(), + usage: stage_usage.clone(), }, &stage_scope, ); diff --git a/lib/components/fabro-workflow/src/handler/llm/pebble.rs b/lib/components/fabro-workflow/src/handler/llm/pebble.rs index 60cd98bab..65b7b143e 100644 --- a/lib/components/fabro-workflow/src/handler/llm/pebble.rs +++ b/lib/components/fabro-workflow/src/handler/llm/pebble.rs @@ -24,12 +24,12 @@ use fabro_mcp::pebble::pebble_servers; use fabro_sandbox::{RunSandbox, SecretRedactor}; use fabro_types::settings::run::RunModelControls; use fabro_types::{ - AgentProfileKind, BilledModelUsage, ModelRef, PermissionLevel, SessionCapability, StageId, - StageTiming, UsdMicros, billing, + AgentProfileKind, ModelRef, ModelUsage, PermissionLevel, SessionCapability, StageId, + StageTiming, }; use fabro_util::home::Home; use lithos_llm::catalog::{ModelId, ProviderId}; -use lithos_llm::types::{Message as LlmMessage, Role, TokenCounts}; +use lithos_llm::types::{Message as LlmMessage, Role, Usage}; use pebble_agent::ToolMiddleware; use pebble_coding_agent::environment::Environment; use pebble_coding_agent::events::{CodingAgentEvent, EventSink, EventSinkError}; @@ -63,7 +63,7 @@ use crate::context::keys::Fidelity; use crate::error::Error; use crate::event::{Emitter, Event, StageScope}; use crate::model_fallback::{ModelFallbackNotice, ModelFallbackPolicy}; -use crate::outcome::{Outcome, billed_model_usage_from_llm}; +use crate::outcome::{Outcome, model_usage_from_llm, with_reported_cost}; use crate::services::FabroRunToolServices; use crate::steering_hub::SteeringHub; use crate::web_search::{self, SearchSecrets}; @@ -174,7 +174,7 @@ struct WorkflowEventSink { scope: StageScope, /// Pebble's fold of every event this sink recorded: the stage's one /// account of what its agent and subagents spent, wrote, and ran. The - /// store folds the same events the same way, so the stage's billing at + /// store folds the same events the same way, so the stage's usage at /// its end is the usage the run showed live. projection: Mutex, } @@ -270,7 +270,6 @@ impl LiveAgent { original_turns = compaction.original_turn_count, preserved_turns = compaction.preserved_turn_count, usage = ?compaction.usage, - cost_usd_micros = ?compaction.cost_usd_micros, "agent stage compacted its conversation" ); } @@ -320,7 +319,7 @@ impl LiveAgent { } } -/// The route as billing names it: provider, model, and the speed tier the +/// The route as usage names it: provider, model, and the speed tier the /// stage asked for. fn route_model(route: &LlmRoute) -> ModelRef { ModelRef::new( @@ -330,62 +329,50 @@ fn route_model(route: &LlmRoute) -> ModelRef { .with_speed(route.controls.speed) } -/// A stage's billing from its account: the whole tree under the root's +/// A stage's usage from its account: the whole tree under the root's /// route, and the rows that split it by model. -struct StageBilling { - total: BilledModelUsage, - by_model: Vec, +struct StageUsage { + total: ModelUsage, + by_model: Vec, } -/// Bills the stage's account from the catalog: the root session at +/// Prices the stage's account from the catalog: the root session at /// `root_model`, its route, and each descendant at its own route where the /// catalog knows it and at the root's otherwise, so a subagent on a cheaper /// or dearer model is priced as what it ran. A descendant on the root's /// route joins the root's row. Where pebble carried a provider-reported -/// cost, that cost stands in for the catalog's estimate. -fn stage_billing( +/// cost, that cost stands in for the catalog's estimate. The total's cost is +/// the rows' sum, which is `None` once a row that used tokens has no cost. +fn stage_usage( catalog: &Catalog, root_model: &ModelRef, account: &SessionProjection, -) -> Result { - let mut groups: Vec<(ModelRef, TokenCounts, Option)> = vec![( - root_model.clone(), - TokenCounts::from(account.usage), - account.cost_usd_micros, - )]; +) -> Result { + // Each group's usage is the sum of pebble's accounts, so its cost is what + // the provider reported, or `None` once an unpriced account is in it. + let mut groups: Vec<(ModelRef, Usage)> = vec![(root_model.clone(), account.usage)]; for descendant in account.descendants.values() { let model = descendant_model(catalog, root_model, descendant); - match groups.iter_mut().find(|(grouped, _, _)| *grouped == model) { - Some((_, tokens, cost)) => { - billing::add_usage(tokens, TokenCounts::from(descendant.usage)); - add_reported_cost(cost, descendant.cost_usd_micros); - } - None => groups.push(( - model, - TokenCounts::from(descendant.usage), - descendant.cost_usd_micros, - )), + match groups.iter_mut().find(|(grouped, _)| *grouped == model) { + Some((_, usage)) => *usage = usage.saturating_add(descendant.usage), + None => groups.push((model, descendant.usage)), } } // The root's row first, then the others by model. groups[1..].sort_by(|left, right| left.0.sort_key().cmp(&right.0.sort_key())); let mut by_model = Vec::with_capacity(groups.len()); - let mut total_tokens = TokenCounts::default(); - let mut total_cost = None; - for (model, tokens, reported) in groups { - let row = billed_model_usage_from_llm(catalog, &model, tokens)? - .with_reported_cost(reported.map(usd_micros)); - billing::add_usage(&mut total_tokens, row.tokens); - UsdMicros::accumulate(&mut total_cost, row.total_usd_micros.map(UsdMicros)); + let mut total = Usage::default(); + for (model, usage) in groups { + let row = with_reported_cost( + model_usage_from_llm(catalog, &model, usage.tokens)?, + usage.cost, + ); + total = total.saturating_add(row.usage); by_model.push(row); } - Ok(StageBilling { - total: BilledModelUsage { - model: root_model.clone(), - tokens: total_tokens, - total_usd_micros: total_cost.map(|cost| cost.0), - }, + Ok(StageUsage { + total: ModelUsage::new(root_model.clone(), total), by_model, }) } @@ -414,17 +401,6 @@ fn descendant_model( ModelRef::new(ProviderId::new(provider), ModelId::new(model)) } -/// Folds a reported cost into a total that stays `None` until one is seen. -fn add_reported_cost(total: &mut Option, cost: Option) { - if let Some(cost) = cost { - *total = Some(total.unwrap_or(0).saturating_add(cost)); - } -} - -fn usd_micros(micros: u64) -> UsdMicros { - UsdMicros(i64::try_from(micros).unwrap_or(i64::MAX)) -} - /// Everything one stage binds to an agent it builds or resumes. struct StageBindings<'a> { node_id: &'a str, @@ -780,24 +756,24 @@ impl PebbleBackend { /// The failed outcome of an agent stage that spent before it failed: the /// failure itself, with the session tree's usage, the files it wrote, and - /// its active time, so the run bills what the stage spent. A billing the + /// its active time, so the run records what the stage spent. A usage the /// catalog cannot price is logged and left off. fn failed_outcome(&self, error: &Error, live: &LiveAgent, plan: &FallbackPlan) -> Outcome { let mut outcome = error.to_fail_outcome(); let account = live.account(); - match stage_billing( + match stage_usage( self.catalog.as_ref(), &route_model(plan.current()), &account, ) { - Ok(billing) => { - outcome.usage = Some(billing.total); - outcome.usage_by_model = billing.by_model; + Ok(usage) => { + outcome.usage = Some(usage.total); + outcome.usage_by_model = usage.by_model; } - Err(billing_error) => { + Err(usage_error) => { tracing::debug!( - error = %billing_error, - "failed agent stage could not be billed" + error = %usage_error, + "failed agent stage could not be priced" ); } } @@ -1001,8 +977,7 @@ impl CodergenBackend for PebbleBackend { .map(structured_output::prompt_response_format); let mut repair_attempts = 0_i64; let mut previous_validation_error = None; - let mut total_usage = TokenCounts::default(); - let mut total_cost = None; + let mut total_usage = Usage::default(); let mut inference_duration = Duration::ZERO; loop { @@ -1026,11 +1001,7 @@ impl CodergenBackend for PebbleBackend { .await; inference_duration = inference_duration.saturating_add(inference_start.elapsed()); let completion = completion_result?; - billing::add_usage(&mut total_usage, completion.response.usage); - UsdMicros::accumulate( - &mut total_cost, - completion.response.cost.as_ref().map(UsdMicros::from_cost), - ); + total_usage = total_usage.saturating_add(completion.response.usage_with_cost()); let response_text = completion.response.text(); let validation_error = if let Some(schema) = &output_schema { @@ -1057,9 +1028,12 @@ impl CodergenBackend for PebbleBackend { continue; } - let stage_usage = - billed_model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage)? - .with_reported_cost(total_cost); + // The provider's own cost, when every answer carried one, stands in + // for the catalog's estimate. + let stage_usage = with_reported_cost( + model_usage_from_llm(self.catalog.as_ref(), &completion.model, total_usage.tokens)?, + total_usage.cost, + ); return Ok(CodergenResult::Text { text: response_text, @@ -1260,7 +1234,7 @@ impl CodergenBackend for PebbleBackend { }; let account = live.account(); - let billing = stage_billing( + let usage = stage_usage( self.catalog.as_ref(), &route_model(fallback_plan.current()), &account, @@ -1288,8 +1262,8 @@ impl CodergenBackend for PebbleBackend { Ok(CodergenResult::Text { text: response, - usage: Some(billing.total), - usage_by_model: billing.by_model, + usage: Some(usage.total), + usage_by_model: usage.by_model, files_touched: account.files_touched, last_file_touched: account.last_file_touched, timing: StageTiming::active_only( @@ -1306,7 +1280,10 @@ mod tests { use fabro_llm::test_support::test_catalog; use lithos_llm::catalog::builtin; - use pebble_coding_agent::events::{CodingAgentEvent, CodingEvent, InputSource, TokenUsage}; + use lithos_llm::types::TokenCounts; + use pebble_coding_agent::events::{ + CodingAgentEvent, CodingEvent, Cost, CostSource, InputSource, Usage, + }; use super::*; @@ -1330,13 +1307,17 @@ mod tests { CodingEvent::AssistantMessage { text: "ok".to_string(), model: model.to_string(), - usage: TokenUsage { - input, - output, - ..TokenUsage::default() + usage: Usage { + tokens: TokenCounts { + input, + output, + ..TokenCounts::default() + }, + cost: cost.map(|usd_micros| Cost { + usd_micros, + source: CostSource::Provider, + }), }, - cost_usd_micros: cost, - cost_source: None, tool_call_count: 0, context_window: None, reasoning: None, @@ -1354,7 +1335,7 @@ mod tests { } #[test] - fn stage_billing_prices_the_root_at_its_route_and_each_descendant_at_its_own() { + fn stage_usage_prices_the_root_at_its_route_and_each_descendant_at_its_own() { let catalog = test_catalog(); let account = account(&[ root(started("openai", "gpt-5.4")), @@ -1376,47 +1357,56 @@ mod tests { root(CodingEvent::ProcessingEnd), ]); - let billing = stage_billing(&catalog, &root_model(), &account).unwrap(); + let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); - assert_eq!(billing.by_model.len(), 2, "{:?}", billing.by_model); - let root_row = &billing.by_model[0]; + assert_eq!(usage.by_model.len(), 2, "{:?}", usage.by_model); + let root_row = &usage.by_model[0]; assert_eq!(root_row.model, root_model()); assert_eq!( - root_row.tokens.input, 111_000, + root_row.usage.tokens.input, 111_000, "the root, the same-route child, and the unknown-route child" ); - assert_eq!(root_row.tokens.output, 26_100); + assert_eq!(root_row.usage.tokens.output, 26_100); let root_priced = - billed_model_usage_from_llm(&catalog, &root_model(), root_row.tokens).unwrap(); - assert_eq!(root_row.total_usd_micros, root_priced.total_usd_micros); + model_usage_from_llm(&catalog, &root_model(), root_row.usage.tokens).unwrap(); + assert_eq!(root_row.usage.cost, root_priced.usage.cost); + assert_eq!( + root_row.usage.cost.map(|cost| cost.source), + Some(CostSource::Catalog) + ); let other_model = ModelRef::new( ProviderId::new("anthropic"), ModelId::new("claude-sonnet-5"), ); - let other_row = &billing.by_model[1]; + let other_row = &usage.by_model[1]; assert_eq!(other_row.model, other_model); - assert_eq!(other_row.tokens.input, 20_000); - assert_eq!(other_row.tokens.output, 2_000); + assert_eq!(other_row.usage.tokens.input, 20_000); + assert_eq!(other_row.usage.tokens.output, 2_000); let other_priced = - billed_model_usage_from_llm(&catalog, &other_model, other_row.tokens).unwrap(); - assert_eq!(other_row.total_usd_micros, other_priced.total_usd_micros); + model_usage_from_llm(&catalog, &other_model, other_row.usage.tokens).unwrap(); + assert_eq!(other_row.usage.cost, other_priced.usage.cost); assert_ne!( - other_row.total_usd_micros, - billed_model_usage_from_llm(&catalog, &root_model(), other_row.tokens) + other_row.usage.cost, + model_usage_from_llm(&catalog, &root_model(), other_row.usage.tokens) .unwrap() - .total_usd_micros, + .usage + .cost, "priced at its own rate, not the root's" ); // The total is the tree's tokens under the root's route, at the rows' summed - // cost. - assert_eq!(billing.total.model, root_model()); - assert_eq!(billing.total.tokens.input, 131_000); - assert_eq!(billing.total.tokens.output, 28_100); + // cost, from the catalog like every row. + assert_eq!(usage.total.model, root_model()); + assert_eq!(usage.total.usage.tokens.input, 131_000); + assert_eq!(usage.total.usage.tokens.output, 28_100); assert_eq!( - billing.total.total_usd_micros, - Some(root_priced.total_usd_micros.unwrap() + other_priced.total_usd_micros.unwrap()) + usage.total.usage.cost, + Some(Cost { + usd_micros: root_priced.usage.cost.unwrap().usd_micros + + other_priced.usage.cost.unwrap().usd_micros, + source: CostSource::Catalog, + }) ); } @@ -1430,22 +1420,29 @@ mod tests { child("ses_child", message("claude-sonnet-5", 500, 50, None)), ]); - let billing = stage_billing(&catalog, &root_model(), &account).unwrap(); + let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); - assert_eq!(billing.by_model[0].total_usd_micros, Some(4_321)); - let child_priced = billed_model_usage_from_llm( + assert_eq!( + usage.by_model[0].usage.cost, + Some(Cost { + usd_micros: 4_321, + source: CostSource::Provider, + }) + ); + let child_priced = model_usage_from_llm( &catalog, - &billing.by_model[1].model, - billing.by_model[1].tokens, + &usage.by_model[1].model, + usage.by_model[1].usage.tokens, ) .unwrap(); + assert_eq!(usage.by_model[1].usage.cost, child_priced.usage.cost); + // One row reported, one priced: the sum is the application's. assert_eq!( - billing.by_model[1].total_usd_micros, - child_priced.total_usd_micros - ); - assert_eq!( - billing.total.total_usd_micros, - Some(4_321 + child_priced.total_usd_micros.unwrap()) + usage.total.usage.cost, + Some(Cost { + usd_micros: 4_321 + child_priced.usage.cost.unwrap().usd_micros, + source: CostSource::Application, + }) ); } @@ -1459,14 +1456,14 @@ mod tests { message("gpt-5.4-mini", 1_000, 100, None), )); - let billing = stage_billing(&catalog, &root_model(), &account).unwrap(); + let usage = stage_usage(&catalog, &root_model(), &account).unwrap(); - let child_row = billing + let child_row = usage .by_model .iter() .find(|row| row.model.model_id.as_str() == "gpt-5.4-mini") .expect("the child is billed as its answers' model on the root's provider"); assert_eq!(child_row.model.provider, root_model().provider); - assert_eq!(child_row.tokens.input, 1_000); + assert_eq!(child_row.usage.tokens.input, 1_000); } } diff --git a/lib/components/fabro-workflow/src/handler/llm/preamble.rs b/lib/components/fabro-workflow/src/handler/llm/preamble.rs index e69e5771d..1fade571b 100644 --- a/lib/components/fabro-workflow/src/handler/llm/preamble.rs +++ b/lib/components/fabro-workflow/src/handler/llm/preamble.rs @@ -594,10 +594,10 @@ mod tests { use lithos_llm::types::TokenCounts; use super::*; - use crate::outcome::{BilledModelUsage, billed_model_usage_from_llm}; + use crate::outcome::{ModelUsage, model_usage_from_llm}; - fn stage_usage(model: &str, input: u64, output: u64) -> BilledModelUsage { - billed_model_usage_from_llm( + fn stage_usage(model: &str, input: u64, output: u64) -> ModelUsage { + model_usage_from_llm( &fabro_llm::test_support::test_catalog(), &ModelRef::new(builtin::anthropic(), ModelId::new(model)), TokenCounts { diff --git a/lib/components/fabro-workflow/src/handler/prompt.rs b/lib/components/fabro-workflow/src/handler/prompt.rs index 3666869d2..12e9d9393 100644 --- a/lib/components/fabro-workflow/src/handler/prompt.rs +++ b/lib/components/fabro-workflow/src/handler/prompt.rs @@ -150,7 +150,7 @@ impl Handler for PromptHandler { response: response_text.clone(), model: response_model, provider: response_provider, - billing: stage_usage.clone(), + usage: stage_usage.clone(), }, &stage_scope, ); diff --git a/lib/components/fabro-workflow/src/lib.rs b/lib/components/fabro-workflow/src/lib.rs index 94f822240..8b06867a0 100644 --- a/lib/components/fabro-workflow/src/lib.rs +++ b/lib/components/fabro-workflow/src/lib.rs @@ -61,7 +61,7 @@ pub fn extract_stage_timings_by_stage_id( timings } -/// Sum of timing in each node across every visit. Use for billing/usage +/// Sum of timing in each node across every visit. Use for usage /// where a retried node should count its full time. `wall_time_ms`, /// `inference_time_ms`, `tool_time_ms`, and `active_time_ms` are all summed /// per node. @@ -125,8 +125,8 @@ mod duration_tests { status: StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: vec![], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: vec![], @@ -159,12 +159,12 @@ mod duration_tests { tool_call_id: None, actor: None, body: EventBody::StageFailed(StageFailedProps { - index: 0, - failure: None, - will_retry: true, - timing: StageTiming::wall_only(wall_time_ms), - billing_by_model: Vec::new(), - billing: None, + index: 0, + failure: None, + will_retry: true, + timing: StageTiming::wall_only(wall_time_ms), + usage_by_model: Vec::new(), + usage: None, }), }; EventEnvelope { seq, event } @@ -252,8 +252,8 @@ mod duration_tests { status: StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: vec![], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: vec![], @@ -289,7 +289,6 @@ pub mod agent_memory; pub mod artifact; pub mod artifact_snapshot; pub mod artifact_upload; -pub mod billing_rollup; pub mod command_log; pub(crate) mod condition; pub mod context; @@ -319,14 +318,15 @@ mod retry; pub mod run_control; pub(crate) mod run_dir; pub mod run_lookup; +pub mod usage_rollup; -pub use billing_rollup::{ - ProjectionBillingByModel, ProjectionBillingRollup, ProjectionBillingStage, - billing_rollup_from_projection, -}; pub use error::{Error, FailureCategory, FailureSignature, FailureSignatureExt, Result}; pub use fabro_types::ManifestPath; pub use steering_hub::{PairControlError, SteeringHub}; +pub use usage_rollup::{ + ProjectionUsageByModel, ProjectionUsageRollup, ProjectionUsageStage, + usage_rollup_from_projection, +}; pub mod run_materialization; pub mod run_options; pub mod run_status; diff --git a/lib/components/fabro-workflow/src/lifecycle/artifact.rs b/lib/components/fabro-workflow/src/lifecycle/artifact.rs index 2d0383485..4a9692222 100644 --- a/lib/components/fabro-workflow/src/lifecycle/artifact.rs +++ b/lib/components/fabro-workflow/src/lifecycle/artifact.rs @@ -23,12 +23,12 @@ use crate::artifact_upload::ArtifactSink; use crate::event::{Emitter, Event, RunNoticeCode, RunNoticeLevel}; use crate::graph::{WorkflowGraph, WorkflowNode}; use crate::lifecycle::event::stage_scope_for; -use crate::outcome::BilledModelUsage; +use crate::outcome::ModelUsage; use crate::runtime_store::RunStoreHandle; use crate::stage_execution::StageExecutionTracker; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; type ArtifactIdentity = (String, String); const ARTIFACT_UPLOAD_RETRY_DELAYS: [Duration; 3] = [ diff --git a/lib/components/fabro-workflow/src/lifecycle/circuit_breaker.rs b/lib/components/fabro-workflow/src/lifecycle/circuit_breaker.rs index 0b29fee43..10b73050c 100644 --- a/lib/components/fabro-workflow/src/lifecycle/circuit_breaker.rs +++ b/lib/components/fabro-workflow/src/lifecycle/circuit_breaker.rs @@ -9,10 +9,10 @@ use fabro_core::state::ExecutionState; use crate::error::{FailureCategory, FailureSignature, FailureSignatureExt}; use crate::graph::{WorkflowGraph, WorkflowNode}; -use crate::outcome::{BilledModelUsage, OutcomeExt}; +use crate::outcome::{ModelUsage, OutcomeExt}; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; /// Sub-lifecycle responsible for tracking failure signatures and tripping the /// circuit breaker when deterministic failure cycles are detected. diff --git a/lib/components/fabro-workflow/src/lifecycle/event.rs b/lib/components/fabro-workflow/src/lifecycle/event.rs index cf0bda9c3..e349ec980 100644 --- a/lib/components/fabro-workflow/src/lifecycle/event.rs +++ b/lib/components/fabro-workflow/src/lifecycle/event.rs @@ -17,12 +17,12 @@ use super::git::GitCheckpointResult; use crate::context::{Context, WorkflowContext}; use crate::event::{Emitter, Event, StageScope}; use crate::graph::{WorkflowGraph, WorkflowNode}; -use crate::outcome::{BilledModelUsage, FailureCategory, FailureDetail, Outcome, StageOutcome}; +use crate::outcome::{FailureCategory, FailureDetail, ModelUsage, Outcome, StageOutcome}; use crate::stage_execution::{StageExecution, StageExecutionTracker}; use crate::{artifact, context}; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; type FailureSignatureSnapshot = ( Option>, Option>, @@ -215,8 +215,8 @@ impl RunLifecycle for EventLifecycle { status: StageOutcome::Succeeded.to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), @@ -241,7 +241,7 @@ impl RunLifecycle for EventLifecycle { &self, ctx: &AttemptContext<'_, WorkflowGraph>, state: &WfRunState, - ) -> CoreResult>> { + ) -> CoreResult>> { let gv = ctx.node.inner(); let execution = self.stage_executions.active(&gv.id); let scope = stage_scope_from_execution(execution.as_deref(), state, &gv.id); @@ -290,8 +290,8 @@ impl RunLifecycle for EventLifecycle { failure, will_retry: true, timing, - billing: outcome.usage.clone(), - billing_by_model: outcome.usage_by_model.clone(), + usage: outcome.usage.clone(), + usage_by_model: outcome.usage_by_model.clone(), actor, }, &scope, @@ -343,8 +343,8 @@ impl RunLifecycle for EventLifecycle { failure, will_retry: false, timing, - billing: outcome.usage.clone(), - billing_by_model: outcome.usage_by_model.clone(), + usage: outcome.usage.clone(), + usage_by_model: outcome.usage_by_model.clone(), actor, }, &scope, @@ -359,8 +359,8 @@ impl RunLifecycle for EventLifecycle { status: outcome.status.to_string(), preferred_label: outcome.preferred_label.clone(), suggested_next_ids: outcome.suggested_next_ids.clone(), - billing: outcome.usage.clone(), - billing_by_model: outcome.usage_by_model.clone(), + usage: outcome.usage.clone(), + usage_by_model: outcome.usage_by_model.clone(), failure: outcome.failure.clone(), notes: outcome.notes.clone(), files_touched: outcome.files_touched.clone(), diff --git a/lib/components/fabro-workflow/src/lifecycle/fidelity.rs b/lib/components/fabro-workflow/src/lifecycle/fidelity.rs index ec8e24107..854763bca 100644 --- a/lib/components/fabro-workflow/src/lifecycle/fidelity.rs +++ b/lib/components/fabro-workflow/src/lifecycle/fidelity.rs @@ -14,11 +14,11 @@ use crate::artifact; use crate::context::{Context, ParallelBranchPreamble, keys}; use crate::graph::{WorkflowGraph, WorkflowNode}; use crate::handler::llm::preamble::build_preamble; -use crate::outcome::{BilledModelUsage, Outcome}; +use crate::outcome::{ModelUsage, Outcome}; use crate::runtime_store::RunStoreHandle; -type WfRunState = ExecutionState>; -type WfNodeDecision = NodeDecision>; +type WfRunState = ExecutionState>; +type WfNodeDecision = NodeDecision>; /// Graphviz edge captured from edge selection, passed to the next node's /// before_node for fidelity/thread resolution. diff --git a/lib/components/fabro-workflow/src/lifecycle/git.rs b/lib/components/fabro-workflow/src/lifecycle/git.rs index 10055c859..c582a9441 100644 --- a/lib/components/fabro-workflow/src/lifecycle/git.rs +++ b/lib/components/fabro-workflow/src/lifecycle/git.rs @@ -11,7 +11,7 @@ use fabro_types::{DiffSummary, RunId}; use crate::event::{Emitter, Event, RunNoticeCode, RunNoticeLevel}; use crate::graph::{WorkflowGraph, WorkflowNode}; use crate::lifecycle::event::stage_scope_for; -use crate::outcome::BilledModelUsage; +use crate::outcome::ModelUsage; use crate::run_options::RunOptions; use crate::sandbox_git::{ checked_git_checkpoint, git_diff, list_diff_numstat, summarize_diff_numstat, @@ -19,8 +19,8 @@ use crate::sandbox_git::{ use crate::sandbox_git_runtime::SandboxGitRuntime; use crate::stage_execution::StageExecutionTracker; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; /// Result of a git checkpoint operation, shared with EventLifecycle. #[derive(Debug, Clone)] diff --git a/lib/components/fabro-workflow/src/lifecycle/hook.rs b/lib/components/fabro-workflow/src/lifecycle/hook.rs index 73a001b3e..5a641c790 100644 --- a/lib/components/fabro-workflow/src/lifecycle/hook.rs +++ b/lib/components/fabro-workflow/src/lifecycle/hook.rs @@ -13,11 +13,11 @@ use fabro_types::RunId; use crate::graph::{WorkflowGraph, WorkflowNode}; use crate::hook_context::set_hook_node; -use crate::outcome::{BilledModelUsage, Outcome, OutcomeExt, StageOutcome}; +use crate::outcome::{ModelUsage, Outcome, OutcomeExt, StageOutcome}; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; -type WfNodeDecision = NodeDecision>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; +type WfNodeDecision = NodeDecision>; /// Sub-lifecycle responsible for running workflow hooks. pub(crate) struct HookLifecycle { diff --git a/lib/components/fabro-workflow/src/lifecycle/mod.rs b/lib/components/fabro-workflow/src/lifecycle/mod.rs index 168c8115e..701e4d3a3 100644 --- a/lib/components/fabro-workflow/src/lifecycle/mod.rs +++ b/lib/components/fabro-workflow/src/lifecycle/mod.rs @@ -35,7 +35,7 @@ use crate::context; use crate::error::FailureSignature; use crate::event::Emitter; use crate::graph::{WorkflowGraph, WorkflowNode}; -use crate::outcome::{BilledModelUsage, Outcome}; +use crate::outcome::{ModelUsage, Outcome}; use crate::run_control::RunControlState; use crate::run_options::RunOptions; use crate::runtime_store::RunStoreHandle; @@ -43,9 +43,9 @@ use crate::sandbox_git_runtime::SandboxGitRuntime; use crate::services::RunLocations; use crate::stage_execution::StageExecutionTracker; -type WfRunState = ExecutionState>; -type WfNodeResult = NodeResult>; -type WfNodeDecision = NodeDecision>; +type WfRunState = ExecutionState>; +type WfNodeResult = NodeResult>; +type WfNodeDecision = NodeDecision>; /// Orchestrates all sub-lifecycles with explicit per-callback ordering. /// Implements `RunLifecycle` by delegating to focused structs. diff --git a/lib/components/fabro-workflow/src/operations/archive.rs b/lib/components/fabro-workflow/src/operations/archive.rs index 8b2ae1876..64f520b92 100644 --- a/lib/components/fabro-workflow/src/operations/archive.rs +++ b/lib/components/fabro-workflow/src/operations/archive.rs @@ -167,11 +167,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }) .await .unwrap(); diff --git a/lib/components/fabro-workflow/src/operations/fork.rs b/lib/components/fabro-workflow/src/operations/fork.rs index 0c1395bb0..e42867d3c 100644 --- a/lib/components/fabro-workflow/src/operations/fork.rs +++ b/lib/components/fabro-workflow/src/operations/fork.rs @@ -427,8 +427,8 @@ mod tests { status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: Vec::new(), diff --git a/lib/components/fabro-workflow/src/operations/retry.rs b/lib/components/fabro-workflow/src/operations/retry.rs index b938610e8..b2615b2c9 100644 --- a/lib/components/fabro-workflow/src/operations/retry.rs +++ b/lib/components/fabro-workflow/src/operations/retry.rs @@ -248,11 +248,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: fabro_types::SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }) .await .unwrap(); diff --git a/lib/components/fabro-workflow/src/operations/start.rs b/lib/components/fabro-workflow/src/operations/start.rs index bbffa9e69..27e915f93 100644 --- a/lib/components/fabro-workflow/src/operations/start.rs +++ b/lib/components/fabro-workflow/src/operations/start.rs @@ -297,7 +297,7 @@ pub(super) async fn execute_persisted_run( } /// Build a conclusion from the store and emit `run.failed` carrying the -/// rolled-up timing and billing. Shared by the engine-failure terminal path, +/// rolled-up timing and usage. Shared by the engine-failure terminal path, /// the bootstrap/completion drop guards, and `persist_detached_failure`. async fn emit_workflow_run_failed( run_id: RunId, @@ -325,7 +325,7 @@ async fn emit_workflow_run_failed( None, None, None, - conclusion.billing, + conclusion.usage, ); if let Err(err) = append_event_to_sink(event_sink, &run_id, &failure_event).await { let rendered_error = collect_chain(&err).join(": "); @@ -1303,11 +1303,12 @@ mod tests { RunPrepareSettings, }; use fabro_types::{ - BilledModelUsage, GitContext, ManifestPath, RunTarget, StageTiming, WorkflowSettings, - fixtures, test_support, + GitContext, ManifestPath, ModelUsage, RunTarget, StageTiming, WorkflowSettings, fixtures, + test_support, }; use fabro_vault::SecretType; use lithos_llm::catalog::builtin; + use lithos_llm::types::Usage; use object_store::memory::InMemory; use super::*; @@ -2556,7 +2557,7 @@ mod tests { run_store: &fabro_store::RunDatabase, node_id: &str, timing: fabro_types::StageTiming, - billing: Option, + usage: Option, ) { crate::event::append_event(run_store, &fixtures::RUN_1, &Event::StageCompleted { node_id: node_id.to_string(), @@ -2566,8 +2567,8 @@ mod tests { status: StageOutcome::Succeeded.to_string(), preferred_label: None, suggested_next_ids: Vec::new(), - billing, - billing_by_model: Vec::new(), + usage, + usage_by_model: Vec::new(), failure: None, notes: None, files_touched: Vec::new(), @@ -2707,7 +2708,7 @@ mod tests { } #[tokio::test] - async fn persist_terminal_engine_failure_uses_conclusion_timing_and_billing() { + async fn persist_terminal_engine_failure_uses_conclusion_timing_and_usage() { let temp = tempfile::tempdir().unwrap(); let (storage_root, run_dir) = storage_root_and_run_dir(&temp); let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await; @@ -2748,17 +2749,11 @@ mod tests { assert_eq!(conclusion.timing.inference_time_ms, 225); assert_eq!(conclusion.timing.tool_time_ms, 375); assert_eq!(conclusion.timing.active_time_ms, 600); - assert_eq!( - conclusion - .billing - .as_ref() - .map(|billing| billing.total_tokens), - Some(150), - ); + assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(150),); } #[tokio::test] - async fn bootstrap_guard_failure_uses_conclusion_timing_and_billing() { + async fn bootstrap_guard_failure_uses_conclusion_timing_and_usage() { let temp = tempfile::tempdir().unwrap(); let (storage_root, _run_dir) = storage_root_and_run_dir(&temp); let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await; @@ -2787,17 +2782,11 @@ mod tests { assert_eq!(conclusion.timing.inference_time_ms, 120); assert_eq!(conclusion.timing.tool_time_ms, 80); assert_eq!(conclusion.timing.active_time_ms, 200); - assert_eq!( - conclusion - .billing - .as_ref() - .map(|billing| billing.total_tokens), - Some(50), - ); + assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(50),); } #[tokio::test] - async fn completion_guard_failure_uses_conclusion_timing_and_billing() { + async fn completion_guard_failure_uses_conclusion_timing_and_usage() { let temp = tempfile::tempdir().unwrap(); let (storage_root, _run_dir) = storage_root_and_run_dir(&temp); let (_persisted, store) = persisted_workflow(MINIMAL_DOT, &storage_root).await; @@ -2826,13 +2815,7 @@ mod tests { assert_eq!(conclusion.timing.inference_time_ms, 70); assert_eq!(conclusion.timing.tool_time_ms, 30); assert_eq!(conclusion.timing.active_time_ms, 100); - assert_eq!( - conclusion - .billing - .as_ref() - .map(|billing| billing.total_tokens), - Some(25), - ); + assert_eq!(conclusion.usage.map(Usage::total_tokens), Some(25),); } #[tokio::test] @@ -3165,7 +3148,7 @@ mod tests { failure: None, final_git_commit_sha: None, stages: vec![], - billing: None, + usage: None, total_retries: 0, diff: fabro_types::RunDiff::default(), }; @@ -3215,11 +3198,10 @@ mod tests { artifact_count: 0, status: "succeeded".to_string(), reason: crate::run_status::SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }) .await .unwrap(); diff --git a/lib/components/fabro-workflow/src/outcome.rs b/lib/components/fabro-workflow/src/outcome.rs index d456a149c..0a1d0fc22 100644 --- a/lib/components/fabro-workflow/src/outcome.rs +++ b/lib/components/fabro-workflow/src/outcome.rs @@ -2,36 +2,43 @@ pub use fabro_core::outcome::{ FailureCategory, FailureDetail, OutcomeMeta, StageOutcome, StageState, }; use fabro_llm::lithos_catalog::Catalog; -pub use fabro_types::BilledModelUsage; -use fabro_types::{BilledTokenCounts, ModelRef}; -use lithos_llm::types::TokenCounts; +use fabro_types::ModelRef; +pub use fabro_types::ModelUsage; +use lithos_llm::types::{Cost, TokenCounts, Usage}; use crate::error::{Error, FailureSignature, classify_failure_reason}; -pub type Outcome = fabro_core::Outcome>; +pub type Outcome = fabro_core::Outcome>; -/// Bills `usage` on `model` from catalog pricing. +/// Prices `tokens` on `model` from the catalog: the usage carries a +/// [`CostSource::Catalog`](lithos_llm::types::CostSource::Catalog) cost when +/// the catalog has rates for the model, and no cost otherwise. /// /// The provider must be one the catalog knows; a passthrough model on a known -/// provider is billed with no cost, since the catalog has no rates for it. -pub fn billed_model_usage_from_llm( +/// provider is priced with no cost, since the catalog has no rates for it. +pub fn model_usage_from_llm( catalog: &Catalog, model: &ModelRef, - usage: TokenCounts, -) -> Result { + tokens: TokenCounts, +) -> Result { if catalog.enabled_provider(model.provider.as_str()).is_none() { return Err(Error::Precondition(format!( "Provider \"{}\" is not configured", model.provider ))); } - let cost = catalog.estimate_cost(&model.handle(), usage, model.speed); - Ok(BilledModelUsage::new(model.clone(), usage, cost)) + let cost = catalog.estimate_cost(&model.handle(), tokens, model.speed); + Ok(ModelUsage::new(model.clone(), Usage { tokens, cost })) } +/// `usage` with `cost` in place of whatever it carried, when a provider +/// reported one; `None` keeps the usage as it is. #[must_use] -pub fn billed_token_counts_from_llm(usage: TokenCounts) -> BilledTokenCounts { - BilledTokenCounts::from_token_counts(usage, None) +pub fn with_reported_cost(mut usage: ModelUsage, cost: Option) -> ModelUsage { + if let Some(cost) = cost { + usage.usage.cost = Some(cost); + } + usage } pub trait OutcomeExt: Sized { @@ -126,11 +133,11 @@ pub fn format_cost(cost: f64) -> String { mod tests { use fabro_llm::lithos_catalog::Catalog; use fabro_llm::test_support::{test_catalog, test_catalog_with_overlay}; - use fabro_types::{ModelRef, UsdMicros}; + use fabro_types::ModelRef; use lithos_llm::catalog::{ModelId, ProviderId, builtin}; - use lithos_llm::types::{Speed, TokenCounts}; + use lithos_llm::types::{Cost, CostSource, Speed, TokenCounts}; - use super::{OutcomeExt, billed_model_usage_from_llm}; + use super::{OutcomeExt, model_usage_from_llm, with_reported_cost}; fn model_ref(provider: ProviderId, model_id: &str, speed: Option) -> ModelRef { ModelRef::new(provider, ModelId::new(model_id)).with_speed(speed) @@ -141,7 +148,7 @@ mod tests { } #[test] - fn billed_model_usage_from_llm_bills_openai_cached_input_and_reasoning_output() { + fn model_usage_from_llm_prices_openai_cached_input_and_reasoning_output() { // Stay under the 272k long-context tier so the standard rates apply. let usage = TokenCounts { input: 100_000, @@ -150,7 +157,7 @@ mod tests { cache_read: 50_000, ..TokenCounts::default() }; - let billed = billed_model_usage_from_llm( + let billed = model_usage_from_llm( &catalog(), &model_ref(builtin::openai(), "gpt-5.4", None), usage, @@ -158,9 +165,15 @@ mod tests { .unwrap(); // 100k input at $2.50/M + 50k cached at $0.25/M + 30k output at $15/M. - assert_eq!(billed.total_usd_micros, Some(712_500)); - assert_eq!(billed.tokens().output, 25_000); - assert_eq!(billed.tokens().reasoning, 5_000); + assert_eq!( + billed.usage.cost, + Some(Cost { + usd_micros: 712_500, + source: CostSource::Catalog, + }) + ); + assert_eq!(billed.usage.tokens.output, 25_000); + assert_eq!(billed.usage.tokens.reasoning, 5_000); } #[test] @@ -170,15 +183,22 @@ mod tests { output: 7, ..TokenCounts::default() }; - let billed = billed_model_usage_from_llm( - &catalog(), - &model_ref(builtin::openai(), "gpt-5.4", None), - usage, - ) - .unwrap() - .with_reported_cost(Some(UsdMicros(125_000))); + let reported = Cost { + usd_micros: 125_000, + source: CostSource::Provider, + }; + let billed = with_reported_cost( + model_usage_from_llm( + &catalog(), + &model_ref(builtin::openai(), "gpt-5.4", None), + usage, + ) + .unwrap(), + Some(reported), + ); - assert_eq!(billed.total_usd_micros, Some(125_000)); + assert_eq!(billed.usage.cost, Some(reported)); + assert_eq!(billed.usage.tokens, usage); } #[test] @@ -192,7 +212,7 @@ mod tests { } #[test] - fn billed_model_usage_from_llm_bills_anthropic_fast_mode_cache_write_pricing() { + fn model_usage_from_llm_prices_anthropic_fast_mode_cache_write_rates() { let usage = TokenCounts { input: 100_000, output: 10_000, @@ -200,7 +220,7 @@ mod tests { cache_read: 20_000, cache_write: 30_000, }; - let billed = billed_model_usage_from_llm( + let billed = model_usage_from_llm( &catalog(), &model_ref(builtin::anthropic(), "claude-opus-5", Some(Speed::Fast)), usage, @@ -209,11 +229,14 @@ mod tests { // Fast rates: $10/M input, $50/M output (incl. reasoning), $1/M cache // read, $12.50/M cache write. - assert_eq!(billed.total_usd_micros, Some(2_145_000)); + assert_eq!( + billed.usage.cost.map(|cost| cost.usd_micros), + Some(2_145_000) + ); } #[test] - fn billed_model_usage_from_llm_uses_injected_custom_catalog() { + fn model_usage_from_llm_uses_injected_custom_catalog() { let catalog = test_catalog_with_overlay( r#" [providers.proxy] @@ -237,20 +260,23 @@ pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_millio output: 500_000, ..TokenCounts::default() }; - let billed = billed_model_usage_from_llm( + let billed = model_usage_from_llm( &catalog, &model_ref(ProviderId::new("proxy"), "canonical-model", None), usage, ) .unwrap(); - assert_eq!(billed.total_usd_micros, Some(2_000_000)); + assert_eq!( + billed.usage.cost.map(|cost| cost.usd_micros), + Some(2_000_000) + ); assert_eq!(billed.model_id(), "canonical-model"); } #[test] fn passthrough_model_on_known_provider_has_no_cost() { - let billed = billed_model_usage_from_llm( + let billed = model_usage_from_llm( &catalog(), &model_ref(builtin::openai(), "brand-new-model", None), TokenCounts { @@ -260,13 +286,13 @@ pricing = { input_usd_micros_per_million = 1000000, output_usd_micros_per_millio }, ) .unwrap(); - assert_eq!(billed.total_usd_micros, None); - assert_eq!(billed.tokens().input, 10); + assert_eq!(billed.usage.cost, None); + assert_eq!(billed.usage.tokens.input, 10); } #[test] fn unknown_provider_is_a_precondition_failure() { - let error = billed_model_usage_from_llm( + let error = model_usage_from_llm( &catalog(), &model_ref(ProviderId::new("nowhere"), "model", None), TokenCounts::default(), diff --git a/lib/components/fabro-workflow/src/pipeline/finalize.rs b/lib/components/fabro-workflow/src/pipeline/finalize.rs index 901c02def..7ef829f5f 100644 --- a/lib/components/fabro-workflow/src/pipeline/finalize.rs +++ b/lib/components/fabro-workflow/src/pipeline/finalize.rs @@ -1,10 +1,10 @@ use std::sync::Arc; use fabro_hooks::{HookContext, HookEvent}; -use fabro_types::{BilledTokenCounts, DiffSummary, EventBody, RunFailure, RunProjection}; +use fabro_types::{DiffSummary, EventBody, RunFailure, RunProjection}; +use lithos_llm::types::Usage; use super::types::{Concluded, Executed, FinalizeOptions, Finalized, PublishOutcome, Published}; -use crate::billing_rollup; use crate::error::{Error, run_failure_from_error, run_failure_from_outcome_failure}; use crate::event::{Event, RunNoticeCode, RunNoticeLevel}; use crate::outcome::{Outcome, StageOutcome}; @@ -14,6 +14,7 @@ use crate::run_status::{FailureReason, RunStatus, SuccessReason}; use crate::runtime_store::RunStoreHandle; use crate::sandbox_git::{git_diff_with_timeout, list_diff_numstat, summarize_diff_numstat}; use crate::services::RunServices; +use crate::usage_rollup; pub fn classify_engine_result( engine_result: &Result, @@ -74,20 +75,20 @@ fn build_conclusion_from_projection( run_wall_time_ms: u64, final_git_commit_sha: Option, ) -> Conclusion { - let billing = projection - .map(billing_rollup::billing_rollup_from_projection) + let rollup = projection + .map(usage_rollup::usage_rollup_from_projection) .unwrap_or_default(); let (stages, total_retries) = projection - .map(|projection| billing.conclusion_stages(projection)) + .map(|projection| rollup.conclusion_stages(projection)) .unwrap_or_default(); Conclusion { timestamp: chrono::Utc::now(), status, - timing: billing.timing.with_wall_time(run_wall_time_ms), + timing: rollup.timing.with_wall_time(run_wall_time_ms), failure, final_git_commit_sha, stages, - billing: billing.billing_if_present(), + usage: rollup.usage_if_present(), total_retries, diff: fabro_types::RunDiff::default(), } @@ -139,8 +140,8 @@ async fn compute_final_patch( } #[cfg(any(test, feature = "test-support"))] -pub(crate) fn billing_from_projection(projection: &RunProjection) -> Option { - billing_rollup::billing_rollup_from_projection(projection).billing_if_present() +pub(crate) fn usage_from_projection(projection: &RunProjection) -> Option { + usage_rollup::usage_rollup_from_projection(projection).usage_if_present() } pub(crate) fn build_terminal_event( @@ -150,7 +151,7 @@ pub(crate) fn build_terminal_event( final_git_commit_sha: Option, final_patch: Option, diff_summary: Option, - billing: Option, + usage: Option, ) -> Event { let outcome_status = outcome.as_ref().map_or( StageOutcome::Failed { @@ -162,7 +163,6 @@ pub(crate) fn build_terminal_event( if outcome_status == StageOutcome::Succeeded || outcome_status == StageOutcome::PartiallySucceeded { - let total_usd_micros = billing.as_ref().and_then(|b| b.total_usd_micros); return Event::WorkflowRunCompleted { timing, artifact_count, @@ -171,11 +171,10 @@ pub(crate) fn build_terminal_event( StageOutcome::PartiallySucceeded => SuccessReason::PartialSuccess, _ => SuccessReason::Completed, }, - total_usd_micros, final_git_commit_sha, final_patch, diff_summary, - billing, + usage, }; } @@ -196,7 +195,7 @@ pub(crate) fn build_terminal_event( final_git_commit_sha, final_patch, diff_summary, - billing, + usage, } } @@ -311,7 +310,7 @@ pub async fn finalize(published: Published, options: &FinalizeOptions) -> Result conclusion.final_git_commit_sha.clone(), conclusion.diff.patch.clone(), conclusion.diff.summary, - conclusion.billing.clone(), + conclusion.usage, ); services.emitter.emit(&terminal_event); @@ -368,8 +367,8 @@ mod tests { use fabro_sandbox::test_support::MockSandbox; use fabro_store::{Database, RunDatabase, RunProjection}; use fabro_types::{ - BilledTokenCounts, EventBody, RunEvent, RunId, RunSpec, StageCompletion, WorkflowSettings, - first_event_seq, fixtures, test_support, + EventBody, RunEvent, RunId, RunSpec, StageCompletion, WorkflowSettings, first_event_seq, + fixtures, test_support, }; use object_store::memory::InMemory; @@ -687,13 +686,13 @@ mod tests { } #[test] - fn conclusion_billing_sums_retry_visit_usage_from_projection() { + fn conclusion_usage_sums_retry_visit_usage_from_projection() { let mut projection = test_projection(); let failed_usage = test_usage("gpt-old", 100, 10); let success_usage = test_usage("gpt-new", 200, 20); let failed = projection.stage_entry("verify", 1, first_event_seq(1)); failed.timing = Some(fabro_types::StageTiming::wall_only(1200)); - failed.usage = BilledTokenCounts::from_billed_usage(std::slice::from_ref(&failed_usage)); + failed.usage = failed_usage.usage; failed.model = Some(failed_usage.model().clone()); failed.completion = Some(StageCompletion { outcome: StageOutcome::Failed { @@ -705,8 +704,7 @@ mod tests { }); let succeeded = projection.stage_entry("verify", 2, first_event_seq(2)); succeeded.timing = Some(fabro_types::StageTiming::wall_only(800)); - succeeded.usage = - BilledTokenCounts::from_billed_usage(std::slice::from_ref(&success_usage)); + succeeded.usage = success_usage.usage; succeeded.model = Some(success_usage.model().clone()); succeeded.completion = Some(StageCompletion { outcome: StageOutcome::Succeeded, @@ -737,16 +735,17 @@ mod tests { None, ); - assert_eq!(conclusion.billing.as_ref().unwrap().input_tokens, 300); - assert_eq!(conclusion.billing.as_ref().unwrap().output_tokens, 30); - assert_eq!( - conclusion.billing.as_ref().unwrap().total_usd_micros, - Some(330) - ); + let usage = conclusion.usage.unwrap(); + assert_eq!(usage.tokens.input, 300); + assert_eq!(usage.tokens.output, 30); + assert_eq!(usage.cost.map(|cost| cost.usd_micros), Some(330)); assert_eq!(conclusion.stages.len(), 1); assert_eq!(conclusion.stages[0].stage_id, "verify"); assert_eq!(conclusion.stages[0].timing.wall_time_ms, 2000); - assert_eq!(conclusion.stages[0].billing_usd_micros, Some(330)); + assert_eq!( + conclusion.stages[0].usage.cost.map(|cost| cost.usd_micros), + Some(330) + ); assert_eq!(conclusion.stages[0].retries, 1); } diff --git a/lib/components/fabro-workflow/src/pipeline/mod.rs b/lib/components/fabro-workflow/src/pipeline/mod.rs index 178646e14..5a195e0ad 100644 --- a/lib/components/fabro-workflow/src/pipeline/mod.rs +++ b/lib/components/fabro-workflow/src/pipeline/mod.rs @@ -12,7 +12,7 @@ mod validate; pub use execute::execute; pub(crate) use finalize::build_conclusion_from_store; #[cfg(any(test, feature = "test-support"))] -pub(crate) use finalize::{billing_from_projection, build_terminal_event}; +pub(crate) use finalize::{build_terminal_event, usage_from_projection}; pub use finalize::{classify_engine_result, conclude, finalize}; pub use initialize::initialize; pub use parse::parse; diff --git a/lib/components/fabro-workflow/src/pipeline/pull_request.rs b/lib/components/fabro-workflow/src/pipeline/pull_request.rs index 32c46241f..db6915877 100644 --- a/lib/components/fabro-workflow/src/pipeline/pull_request.rs +++ b/lib/components/fabro-workflow/src/pipeline/pull_request.rs @@ -12,7 +12,7 @@ use fabro_types::PullRequestLink; use fabro_types::settings::run::MergeStrategy; use fabro_util::text::strip_goal_decoration; use lithos_llm::catalog::ProviderId; -use lithos_llm::types::{Message, Role}; +use lithos_llm::types::{Cost, Message, Role}; use tokio::time::sleep; use tracing::{debug, info, warn}; @@ -149,9 +149,8 @@ fn truncate_pr_body(body: &str) -> String { } /// Format an optional cost as `$X.XX` or an en-dash when absent. -fn format_cost(cost_usd_micros: Option) -> String { - cost_usd_micros - .map(|value| value as f64 / 1_000_000.0) +fn format_cost(cost: Option) -> String { + cost.map(|cost| cost.usd_micros as f64 / 1_000_000.0) .map_or_else(|| "\u{2013}".to_string(), outcome_format_cost) } @@ -180,7 +179,7 @@ fn format_arc_details_section( // Cost table let total_duration = format_duration_ms(conclusion.timing.wall_time_ms); - let total_cost_str = format_cost(conclusion.billing.as_ref().and_then(|b| b.total_usd_micros)); + let total_cost_str = format_cost(conclusion.usage.and_then(|usage| usage.cost)); let stage_count = conclusion.stages.len(); parts.push(format!( "
\nRan {stage_count} {} in {total_duration} for {total_cost_str}", @@ -192,7 +191,7 @@ fn format_arc_details_section( parts.push("|---|---|---|---|".to_string()); for stage in &conclusion.stages { let dur = format_duration_ms(stage.timing.wall_time_ms); - let cost = format_cost(stage.billing_usd_micros); + let cost = format_cost(stage.usage.cost); parts.push(format!( "| {} | {} | {} | {} |", stage.stage_label, dur, cost, stage.retries @@ -696,13 +695,13 @@ mod tests { use fabro_llm::{Response, ResponseStream}; use fabro_store::Database; use fabro_types::{ - BilledTokenCounts, RunProjection, RunSpec, SuccessReason, WorkflowSettings, - first_event_seq, fixtures, test_support, + RunProjection, RunSpec, SuccessReason, WorkflowSettings, first_event_seq, fixtures, + test_support, }; use fabro_vault::{SecretType, Vault}; use httpmock::Method::{GET, POST}; use httpmock::MockServer; - use lithos_llm::types::{ContentPart, TokenCounts}; + use lithos_llm::types::{ContentPart, CostSource, TokenCounts, Usage}; use object_store::memory::InMemory; use tokio::sync::RwLock as AsyncRwLock; @@ -871,6 +870,17 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr .unwrap() } + /// A usage with only a catalog cost, for the cost table. + fn priced(usd_micros: u64) -> Usage { + Usage { + tokens: TokenCounts::default(), + cost: Some(Cost { + usd_micros, + source: CostSource::Catalog, + }), + } + } + fn make_test_conclusion() -> Conclusion { Conclusion { timestamp: Utc::now(), @@ -880,31 +890,28 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr final_git_commit_sha: None, stages: vec![ StageSummary { - stage_id: "plan".to_string(), - stage_label: "plan".to_string(), - timing: fabro_types::StageTiming::wall_only(45_000), - billing_usd_micros: Some(120_000), - retries: 0, + stage_id: "plan".to_string(), + stage_label: "plan".to_string(), + timing: fabro_types::StageTiming::wall_only(45_000), + usage: priced(120_000), + retries: 0, }, StageSummary { - stage_id: "implement".to_string(), - stage_label: "implement".to_string(), - timing: fabro_types::StageTiming::wall_only(90_000), - billing_usd_micros: Some(250_000), - retries: 0, + stage_id: "implement".to_string(), + stage_label: "implement".to_string(), + timing: fabro_types::StageTiming::wall_only(90_000), + usage: priced(250_000), + retries: 0, }, StageSummary { - stage_id: "simplify".to_string(), - stage_label: "simplify".to_string(), - timing: fabro_types::StageTiming::wall_only(15_000), - billing_usd_micros: Some(50_000), - retries: 0, + stage_id: "simplify".to_string(), + stage_label: "simplify".to_string(), + timing: fabro_types::StageTiming::wall_only(15_000), + usage: priced(50_000), + retries: 0, }, ], - billing: Some(BilledTokenCounts { - total_usd_micros: Some(420_000), - ..BilledTokenCounts::default() - }), + usage: Some(priced(420_000)), total_retries: 0, diff: fabro_types::RunDiff::default(), } @@ -929,9 +936,9 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr fn format_arc_details_no_cost() { let mut conclusion = make_test_conclusion(); for stage in &mut conclusion.stages { - stage.billing_usd_micros = None; + stage.usage.cost = None; } - conclusion.billing = None; + conclusion.usage = None; let section = format_arc_details_section(&conclusion, None, None); // En-dash for missing costs @@ -1217,8 +1224,8 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: vec![], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: vec![], @@ -1639,8 +1646,8 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr status: "succeeded".to_string(), preferred_label: None, suggested_next_ids: vec![], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: None, files_touched: vec![], @@ -1870,13 +1877,12 @@ capabilities = { text = true, tools = true, response_format = { json_object = tr artifact_count: 0, status: "succeeded".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: Some( "diff --git a/src/lib.rs b/src/lib.rs\n+fn from_store() {}\n".to_string(), ), diff_summary: None, - billing: None, + usage: None, }) .await .unwrap(); diff --git a/lib/components/fabro-workflow/src/run_lookup.rs b/lib/components/fabro-workflow/src/run_lookup.rs index fc25cb9e2..d9b3ce48f 100644 --- a/lib/components/fabro-workflow/src/run_lookup.rs +++ b/lib/components/fabro-workflow/src/run_lookup.rs @@ -141,16 +141,15 @@ impl RunInfo { } pub fn total_cost(&self) -> Option { - self.summary - .as_ref() - .and_then(|summary| summary.billing.as_ref()?.total_usd_micros) + self.total_usd_micros() .map(|value| value as f64 / 1_000_000.0) } - pub fn total_usd_micros(&self) -> Option { + pub fn total_usd_micros(&self) -> Option { self.summary .as_ref() - .and_then(|summary| summary.billing.as_ref()?.total_usd_micros) + .and_then(|summary| summary.usage.cost) + .map(|cost| cost.usd_micros) } pub fn source_directory(&self) -> Option<&str> { diff --git a/lib/components/fabro-workflow/src/test_support.rs b/lib/components/fabro-workflow/src/test_support.rs index 8429e2c5d..081027e93 100644 --- a/lib/components/fabro-workflow/src/test_support.rs +++ b/lib/components/fabro-workflow/src/test_support.rs @@ -16,7 +16,7 @@ use fabro_types::ModelRef; #[cfg(feature = "test-support")] use lithos_llm::catalog::ProviderId; use lithos_llm::catalog::{ModelId, builtin}; -use lithos_llm::types::TokenCounts; +use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use object_store::local::LocalFileSystem; use crate::artifact_upload::ArtifactSink; @@ -26,7 +26,7 @@ use crate::handler::HandlerRegistry; use crate::outcome::Outcome; use crate::pipeline; use crate::pipeline::types::{Executed, Initialized}; -use crate::pipeline::{billing_from_projection, build_terminal_event}; +use crate::pipeline::{build_terminal_event, usage_from_projection}; use crate::records::Checkpoint; use crate::run_options::RunOptions; use crate::sandbox_git_runtime::SandboxGitRuntime; @@ -51,7 +51,7 @@ pub(crate) fn test_configured_provider_ids( /// (FINALIZE). /// /// The first flush is needed because `StoreProgressLogger` forwards events -/// through an mpsc channel — without it, billing would read from a stale +/// through an mpsc channel — without it, usage would read from a stale /// checkpoint. The second flush ensures the just-emitted terminal event is /// persisted before tests reopen the run store. async fn execute_and_emit_terminal(initialized: InitializedState) -> Executed { @@ -62,7 +62,7 @@ async fn execute_and_emit_terminal(initialized: InitializedState) -> Executed { .await .expect("test run events should persist"); let state = executed.engine.run.run_store.state().await.ok(); - let billing = state.as_ref().and_then(billing_from_projection); + let usage = state.as_ref().and_then(usage_from_projection); let event = build_terminal_event( &executed.outcome, fabro_types::RunTiming::wall_only(executed.wall_time_ms), @@ -70,7 +70,7 @@ async fn execute_and_emit_terminal(initialized: InitializedState) -> Executed { None, None, None, - billing, + usage, ); executed.engine.run.emitter.emit(&event); initialized @@ -81,25 +81,29 @@ async fn execute_and_emit_terminal(initialized: InitializedState) -> Executed { executed } -/// Construct a fully-populated `BilledModelUsage` for tests. Centralised so -/// callers don't keep rebuilding the same JSON skeleton. +/// Construct a fully-populated `ModelUsage` for tests: `input_tokens` and +/// `output_tokens` on an OpenAI model, priced from the catalog at one micro +/// per token. Centralised so callers don't keep rebuilding the same skeleton. #[must_use] pub fn test_usage( model_id: &str, input_tokens: u64, output_tokens: u64, -) -> fabro_types::BilledModelUsage { - let mut usage = fabro_types::BilledModelUsage::new( +) -> fabro_types::ModelUsage { + fabro_types::ModelUsage::new( ModelRef::new(builtin::openai(), ModelId::new(model_id)), - TokenCounts { - input: input_tokens, - output: output_tokens, - ..TokenCounts::default() + Usage { + tokens: TokenCounts { + input: input_tokens, + output: output_tokens, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: input_tokens.saturating_add(output_tokens), + source: CostSource::Catalog, + }), }, - None, - ); - usage.total_usd_micros = Some(i64::try_from(input_tokens + output_tokens).unwrap_or(i64::MAX)); - usage + ) } /// Append the `RunStartRequested → RunRunnable → RunStarting → RunRunning` diff --git a/lib/components/fabro-workflow/src/billing_rollup.rs b/lib/components/fabro-workflow/src/usage_rollup.rs similarity index 61% rename from lib/components/fabro-workflow/src/billing_rollup.rs rename to lib/components/fabro-workflow/src/usage_rollup.rs index 2c3a56b34..63b19f479 100644 --- a/lib/components/fabro-workflow/src/billing_rollup.rs +++ b/lib/components/fabro-workflow/src/usage_rollup.rs @@ -1,17 +1,18 @@ -pub use fabro_types::billing_rollup::{ - ProjectionBillingByModel, ProjectionBillingRollup, ProjectionBillingStage, - billing_rollup_from_projection, +pub use fabro_types::usage_rollup::{ + ProjectionUsageByModel, ProjectionUsageRollup, ProjectionUsageStage, + usage_rollup_from_projection, }; #[cfg(test)] mod tests { use fabro_types::{ - AttrValue, BilledTokenCounts, Graph, ModelRef, Node, RunProjection, RunSpec, - StageCompletion, StageOutcome, first_event_seq, test_support, + AttrValue, Graph, ModelRef, Node, RunProjection, RunSpec, StageCompletion, StageOutcome, + first_event_seq, test_support, }; use lithos_llm::catalog::{ModelId, builtin}; + use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; - use super::billing_rollup_from_projection; + use super::usage_rollup_from_projection; use crate::test_support::test_usage; fn test_projection() -> RunProjection { @@ -23,15 +24,15 @@ mod tests { } #[test] - fn by_model_splits_a_completed_stage_by_its_billing_rows() { + fn by_model_splits_a_completed_stage_by_its_usage_rows() { let mut projection = test_projection(); let root = test_usage("gpt-root", 100, 10); let child = test_usage("gpt-child", 7, 1); let stage = projection.stage_entry("work", 1, first_event_seq(1)); stage.timing = Some(fabro_types::StageTiming::wall_only(100)); - stage.usage = BilledTokenCounts::from_billed_usage(&[root.clone(), child.clone()]); + stage.usage = root.usage.saturating_add(child.usage); stage.model = Some(root.model().clone()); - stage.billing_by_model = vec![root.clone(), child.clone()]; + stage.usage_by_model = vec![root.clone(), child.clone()]; stage.completion = Some(StageCompletion { outcome: StageOutcome::Succeeded, notes: None, @@ -39,9 +40,9 @@ mod tests { timestamp: chrono::Utc::now(), }); - let rollup = billing_rollup_from_projection(&projection); + let rollup = usage_rollup_from_projection(&projection); - assert_eq!(rollup.totals.input_tokens, 107); + assert_eq!(rollup.totals.tokens.input, 107); assert_eq!(rollup.stages[0].model.as_ref(), Some(root.model())); assert_eq!(rollup.by_model.len(), 2, "{:?}", rollup.by_model); let entry = |model_id: &str| { @@ -52,17 +53,11 @@ mod tests { .unwrap_or_else(|| panic!("a row for {model_id}")) }; assert_eq!(entry("gpt-root").stages, 1); - assert_eq!(entry("gpt-root").billing.input_tokens, 100); - assert_eq!( - entry("gpt-root").billing.total_usd_micros, - root.total_usd_micros - ); + assert_eq!(entry("gpt-root").usage.tokens.input, 100); + assert_eq!(entry("gpt-root").usage.cost, root.usage.cost); assert_eq!(entry("gpt-child").stages, 1); - assert_eq!(entry("gpt-child").billing.input_tokens, 7); - assert_eq!( - entry("gpt-child").billing.total_usd_micros, - child.total_usd_micros - ); + assert_eq!(entry("gpt-child").usage.tokens.input, 7); + assert_eq!(entry("gpt-child").usage.cost, child.usage.cost); } #[test] @@ -72,7 +67,7 @@ mod tests { let success_usage = test_usage("gpt-new", 200, 20); let first = projection.stage_entry("verify", 1, first_event_seq(1)); first.timing = Some(fabro_types::StageTiming::wall_only(1200)); - first.usage = BilledTokenCounts::from_billed_usage(std::slice::from_ref(&failed_usage)); + first.usage = failed_usage.usage; first.model = Some(failed_usage.model().clone()); first.completion = Some(StageCompletion { outcome: StageOutcome::Failed { @@ -84,7 +79,7 @@ mod tests { }); let second = projection.stage_entry("verify", 2, first_event_seq(2)); second.timing = Some(fabro_types::StageTiming::wall_only(800)); - second.usage = BilledTokenCounts::from_billed_usage(std::slice::from_ref(&success_usage)); + second.usage = success_usage.usage; second.model = Some(success_usage.model().clone()); second.completion = Some(StageCompletion { outcome: StageOutcome::Succeeded, @@ -93,7 +88,7 @@ mod tests { timestamp: chrono::Utc::now(), }); - let rollup = billing_rollup_from_projection(&projection); + let rollup = usage_rollup_from_projection(&projection); assert_eq!(rollup.stages.len(), 1); assert_eq!(rollup.stages[0].node_id, "verify"); @@ -105,27 +100,33 @@ mod tests { Some("gpt-new") ); assert_eq!(rollup.stages[0].timing.wall_time_ms, 2000); - assert_eq!(rollup.stages[0].billing.input_tokens, 300); - assert_eq!(rollup.stages[0].billing.output_tokens, 30); - assert_eq!(rollup.stages[0].billing.total_usd_micros, Some(330)); + assert_eq!(rollup.stages[0].usage.tokens.input, 300); + assert_eq!(rollup.stages[0].usage.tokens.output, 30); + assert_eq!( + rollup.stages[0].usage.cost, + Some(Cost { + usd_micros: 330, + source: CostSource::Catalog, + }) + ); assert_eq!(rollup.timing.wall_time_ms, 2000); - assert_eq!(rollup.totals.input_tokens, 300); - assert_eq!(rollup.totals.output_tokens, 30); - assert_eq!(rollup.totals.total_usd_micros, Some(330)); - assert_eq!(rollup.billed_visit_count, 2); + assert_eq!(rollup.totals.tokens.input, 300); + assert_eq!(rollup.totals.tokens.output, 30); + assert_eq!(rollup.totals.cost.map(|cost| cost.usd_micros), Some(330)); + assert_eq!(rollup.usage_visit_count, 2); assert_eq!(rollup.by_model.len(), 2); assert_eq!(rollup.by_model[0].model.model_id.as_str(), "gpt-new"); assert_eq!(rollup.by_model[0].stages, 1); - assert_eq!(rollup.by_model[0].billing.input_tokens, 200); + assert_eq!(rollup.by_model[0].usage.tokens.input, 200); assert_eq!(rollup.by_model[1].model.model_id.as_str(), "gpt-old"); assert_eq!(rollup.by_model[1].stages, 1); - assert_eq!(rollup.by_model[1].billing.input_tokens, 100); + assert_eq!(rollup.by_model[1].usage.tokens.input, 100); } #[test] - fn rollup_includes_completed_non_llm_stage_rows_with_zero_billing() { + fn rollup_includes_completed_non_llm_stage_rows_with_zero_usage() { let mut projection = test_projection(); let stage = projection.stage_entry("build", 1, first_event_seq(1)); stage.timing = Some(fabro_types::StageTiming::wall_only(25)); @@ -136,16 +137,16 @@ mod tests { timestamp: chrono::Utc::now(), }); - let rollup = billing_rollup_from_projection(&projection); + let rollup = usage_rollup_from_projection(&projection); assert_eq!(rollup.stages.len(), 1); assert_eq!(rollup.stages[0].node_id, "build"); assert_eq!(rollup.stages[0].timing.wall_time_ms, 25); assert!(rollup.stages[0].model.is_none()); - assert_eq!(rollup.stages[0].billing.input_tokens, 0); + assert_eq!(rollup.stages[0].usage, Usage::default()); assert_eq!(rollup.timing.wall_time_ms, 25); assert!(rollup.by_model.is_empty()); - assert!(rollup.billing_if_present().is_none()); + assert!(rollup.usage_if_present().is_none()); } #[test] @@ -169,7 +170,7 @@ mod tests { timestamp: chrono::Utc::now(), }); - let rollup = billing_rollup_from_projection(&projection); + let rollup = usage_rollup_from_projection(&projection); assert_eq!(rollup.stages.len(), 0); assert_eq!(rollup.timing.wall_time_ms, 0); @@ -181,26 +182,65 @@ mod tests { let model = ModelRef::new(builtin::openai(), ModelId::new("gpt-5.4")); let stage = projection.stage_entry("agent", 1, first_event_seq(1)); stage.started_at = Some(chrono::Utc::now()); - stage.usage = BilledTokenCounts { - input_tokens: 500_000, - output_tokens: 125_000, - total_tokens: 625_000, - ..BilledTokenCounts::default() - }; + stage.usage = Usage::from(TokenCounts { + input: 500_000, + output: 125_000, + ..TokenCounts::default() + }); stage.model = Some(model.clone()); - let rollup = billing_rollup_from_projection(&projection); + let rollup = usage_rollup_from_projection(&projection); // The rollup keeps the shape of what the events recorded. Costs come // from the events themselves; an in-flight stage that has recorded no // cost yet stays unpriced rather than being re-estimated here. assert_eq!(rollup.stages.len(), 1); assert_eq!(rollup.stages[0].node_id, "agent"); - assert_eq!(rollup.stages[0].billing.total_usd_micros, None); - assert_eq!(rollup.stages[0].billing.input_tokens, 500_000); - assert_eq!(rollup.totals.total_usd_micros, None); + assert_eq!(rollup.stages[0].usage.cost, None); + assert_eq!(rollup.stages[0].usage.tokens.input, 500_000); + assert_eq!(rollup.totals.cost, None); assert_eq!(rollup.by_model.len(), 1); - assert_eq!(rollup.by_model[0].billing.input_tokens, 500_000); + assert_eq!(rollup.by_model[0].usage.tokens.input, 500_000); + } + + #[test] + fn rollup_totals_lose_their_cost_once_an_unpriced_stage_used_tokens() { + let mut projection = test_projection(); + let priced = test_usage("gpt-priced", 100, 10); + let first = projection.stage_entry("plan", 1, first_event_seq(1)); + first.usage = priced.usage; + first.model = Some(priced.model().clone()); + first.completion = Some(StageCompletion { + outcome: StageOutcome::Succeeded, + notes: None, + failure_reason: None, + timestamp: chrono::Utc::now(), + }); + let second = projection.stage_entry("work", 1, first_event_seq(2)); + second.usage = Usage::from(TokenCounts { + input: 5, + ..TokenCounts::default() + }); + second.model = Some(ModelRef::new(builtin::openai(), ModelId::new("mystery"))); + second.completion = Some(StageCompletion { + outcome: StageOutcome::Succeeded, + notes: None, + failure_reason: None, + timestamp: chrono::Utc::now(), + }); + + let rollup = usage_rollup_from_projection(&projection); + + // A total cost is known only when every part is priced; the per-stage + // rows keep their own. + assert_eq!(rollup.totals.tokens.input, 105); + assert_eq!(rollup.totals.cost, None); + assert_eq!(rollup.stages[0].usage.cost, priced.usage.cost); + assert_eq!(rollup.stages[1].usage.cost, None); + assert_eq!( + rollup.usage_if_present().map(|usage| usage.cost), + Some(None) + ); } fn run_spec_with_boundary_nodes() -> RunSpec { diff --git a/lib/components/fabro-workflow/tests/it/integration.rs b/lib/components/fabro-workflow/tests/it/integration.rs index 118fc20ff..6937a551f 100644 --- a/lib/components/fabro-workflow/tests/it/integration.rs +++ b/lib/components/fabro-workflow/tests/it/integration.rs @@ -61,6 +61,7 @@ use fabro_workflow::test_support::{ use fabro_workflow::transforms::stylesheet::{apply_stylesheet, parse_stylesheet}; use fabro_workflow::transforms::{StylesheetApplicationTransform, TemplateTransform, Transform}; use lithos_llm::catalog::ProviderId; +use lithos_llm::types::{Cost, CostSource}; use object_store::local::LocalFileSystem; use tokio_util::sync::CancellationToken; use ulid::Ulid; @@ -2810,7 +2811,7 @@ async fn workflow_persists_authoritative_openrouter_cost_for_agent_stage() { use httpmock::MockServer; const AUTHORITATIVE_COST_USD: f64 = 0.125; - const AUTHORITATIVE_COST_USD_MICROS: i64 = 125_000; + const AUTHORITATIVE_COST_USD_MICROS: u64 = 125_000; let server = MockServer::start_async().await; let text_chunk = serde_json::json!({ @@ -2918,11 +2919,14 @@ enabled = true let work = state .stage(&fabro_types::StageId::new("work", 1)) .expect("agent stage should be projected"); - assert_eq!(work.usage.input_tokens, 11); - assert_eq!(work.usage.output_tokens, 7); + assert_eq!(work.usage.tokens.input, 11); + assert_eq!(work.usage.tokens.output, 7); assert_eq!( - work.usage.total_usd_micros, - Some(AUTHORITATIVE_COST_USD_MICROS), + work.usage.cost, + Some(Cost { + usd_micros: AUTHORITATIVE_COST_USD_MICROS, + source: CostSource::Provider, + }), "provider-reported usage.cost should override the catalog estimate" ); @@ -8286,7 +8290,7 @@ async fn workflow_run_with_vault_only_openai_codex_builds_pr_body() { failure: None, final_git_commit_sha: None, stages: Vec::new(), - billing: None, + usage: None, total_retries: 0, diff: fabro_types::RunDiff::default(), }), diff --git a/lib/components/fabro-workflow/tests/it/pebble_agent.rs b/lib/components/fabro-workflow/tests/it/pebble_agent.rs index 71b712c9b..1842d9e3d 100644 --- a/lib/components/fabro-workflow/tests/it/pebble_agent.rs +++ b/lib/components/fabro-workflow/tests/it/pebble_agent.rs @@ -51,8 +51,8 @@ const MODEL: &str = "mock-model"; const PROVIDER: &str = "mock"; const CHAT_PATH: &str = "/v1/chat/completions"; const TOOL_RESULT_MARKER: &str = r#""role":"tool""#; -const INPUT_TOKENS_PER_CALL: i64 = 11; -const OUTPUT_TOKENS_PER_CALL: i64 = 7; +const INPUT_TOKENS_PER_CALL: u64 = 11; +const OUTPUT_TOKENS_PER_CALL: u64 = 7; // --- Scripted model --------------------------------------------------------- @@ -396,17 +396,17 @@ async fn write_file_under_profile(profile: &str, tool: &str, path_key: &str) { let work = work_stage(&state); assert_eq!(work.response.as_deref(), Some("Done"), "{profile}"); assert_eq!( - work.usage.input_tokens, + work.usage.tokens.input, 2 * INPUT_TOKENS_PER_CALL, "{profile}: two model calls of input" ); assert_eq!( - work.usage.output_tokens, + work.usage.tokens.output, 2 * OUTPUT_TOKENS_PER_CALL, "{profile}" ); assert_eq!( - work.usage.total_usd_micros, + work.usage.cost.map(|cost| cost.usd_micros), Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), "{profile}: cost from the catalog's pricing" ); @@ -823,25 +823,20 @@ async fn a_stage_that_fails_after_spending_bills_what_it_spent() { } ); assert_eq!( - work.usage.input_tokens, + work.usage.tokens.input, 2 * INPUT_TOKENS_PER_CALL, "the two answered calls are billed" ); - assert_eq!(work.usage.output_tokens, 2 * OUTPUT_TOKENS_PER_CALL); + assert_eq!(work.usage.tokens.output, 2 * OUTPUT_TOKENS_PER_CALL); assert_eq!( - work.usage.total_usd_micros, + work.usage.cost.map(|cost| cost.usd_micros), Some(2 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), "priced from the catalog like a completed stage" ); + assert_eq!(work.usage_by_model.len(), 1, "{:?}", work.usage_by_model); assert_eq!( - work.billing_by_model.len(), - 1, - "{:?}", - work.billing_by_model - ); - assert_eq!( - work.billing_by_model[0].tokens.input, - u64::try_from(2 * INPUT_TOKENS_PER_CALL).unwrap() + work.usage_by_model[0].usage.tokens.input, + 2 * INPUT_TOKENS_PER_CALL ); assert!( tokio::fs::try_exists(&second).await.unwrap(), @@ -862,12 +857,9 @@ async fn a_stage_that_fails_after_spending_bills_what_it_spent() { panic!("stage.failed carries its props: {failed:?}"); }; assert!(!props.will_retry); - let billing = props.billing.as_ref().expect("the failed stage is billed"); - assert_eq!( - billing.tokens.input, - u64::try_from(2 * INPUT_TOKENS_PER_CALL).unwrap() - ); - assert_eq!(props.billing_by_model, vec![billing.clone()]); + let usage = props.usage.as_ref().expect("the failed stage is priced"); + assert_eq!(usage.usage.tokens.input, 2 * INPUT_TOKENS_PER_CALL); + assert_eq!(props.usage_by_model, vec![usage.clone()]); } // --- Questions, subagents, MCP @@ -1006,13 +998,13 @@ async fn a_subagent_runs_under_its_parent_session() { // child's one. let work = work_stage(&state); assert_eq!( - work.usage.input_tokens, + work.usage.tokens.input, 4 * INPUT_TOKENS_PER_CALL, "the child's call is the stage's too" ); - assert_eq!(work.usage.output_tokens, 4 * OUTPUT_TOKENS_PER_CALL); + assert_eq!(work.usage.tokens.output, 4 * OUTPUT_TOKENS_PER_CALL); assert_eq!( - work.usage.total_usd_micros, + work.usage.cost.map(|cost| cost.usd_micros), Some(4 * (INPUT_TOKENS_PER_CALL + 2 * OUTPUT_TOKENS_PER_CALL)), "priced from the catalog for every call" ); @@ -1020,37 +1012,26 @@ async fn a_subagent_runs_under_its_parent_session() { .agent .as_ref() .expect("the stage carries pebble's fold"); - let (descendants, _) = agent.descendant_usage(); + let descendants = agent.descendant_usage(); assert_eq!( - u64::try_from(work.usage.input_tokens).unwrap(), - agent.usage.input + descendants.input, + work.usage.tokens.input, + agent.usage.tokens.input + descendants.tokens.input, "the completed usage is what the live fold showed" ); - assert_eq!( - descendants.input, - u64::try_from(INPUT_TOKENS_PER_CALL).unwrap() - ); + assert_eq!(descendants.tokens.input, INPUT_TOKENS_PER_CALL); // The child ran on its parent's model, so the split is one row carrying // the tree. + assert_eq!(work.usage_by_model.len(), 1, "{:?}", work.usage_by_model); assert_eq!( - work.billing_by_model.len(), - 1, - "{:?}", - work.billing_by_model + work.usage_by_model[0].usage.tokens.input, + 4 * INPUT_TOKENS_PER_CALL ); assert_eq!( - work.billing_by_model[0].tokens.input, - u64::try_from(4 * INPUT_TOKENS_PER_CALL).unwrap() - ); - assert_eq!( - Some(&work.billing_by_model[0].model), + Some(&work.usage_by_model[0].model), work.model.as_ref(), "billed under the root's route" ); - assert_eq!( - work.billing_by_model[0].total_usd_micros, - work.usage.total_usd_micros - ); + assert_eq!(work.usage_by_model[0].usage.cost, work.usage.cost); let agent_events = coding_events(&stage.events); let root_session = agent_events diff --git a/lib/foundation/fabro-api/build.rs b/lib/foundation/fabro-api/build.rs index 5e70009fd..1834074df 100644 --- a/lib/foundation/fabro-api/build.rs +++ b/lib/foundation/fabro-api/build.rs @@ -384,7 +384,7 @@ fn main() { &[], ), ("StageProjection", "fabro_types::StageProjection", &[]), - ("BilledModelUsage", "fabro_types::BilledModelUsage", &[]), + ("ModelUsage", "fabro_types::ModelUsage", &[]), ( "StageInferenceProjection", "fabro_types::StageInferenceProjection", @@ -497,7 +497,6 @@ fn main() { "pebble_coding_agent::projection::PromptDelta", &[], ), - ("TokenUsage", "pebble_coding_agent::events::TokenUsage", &[]), ( "McpToolSummary", "pebble_coding_agent::events::McpToolSummary", @@ -611,10 +610,10 @@ fn main() { "fabro_types::PendingInterviewRecord", &[], ), - ("CompletionUsage", "lithos_llm::types::TokenCounts", &[]), - ("BilledTokenCounts", "fabro_types::BilledTokenCounts", &[]), - ("BillingModelRef", "fabro_types::ModelRef", &[]), - ("BillingSpeed", "lithos_llm::types::Speed", &[]), + ("TokenCounts", "lithos_llm::types::TokenCounts", &[]), + ("Usage", "lithos_llm::types::Usage", &[]), + ("UsageModelRef", "fabro_types::ModelRef", &[]), + ("Speed", "lithos_llm::types::Speed", &[]), ("ExecOutputTail", "fabro_types::ExecOutputTail", &[]), ("StageTiming", "fabro_types::StageTiming", &[]), ("RunTiming", "fabro_types::RunTiming", &[]), @@ -852,7 +851,7 @@ fn main() { "lithos_llm::types::ResponseFormat", &[], ), - ("CompletionCost", "lithos_llm::types::Cost", &[]), + ("Cost", "lithos_llm::types::Cost", &[]), ("WorkflowVersion", "fabro_types::WorkflowVersion", &[]), ("RunIntent", "fabro_types::RunIntent", &[]), ("RunIntentArgs", "fabro_types::RunIntentArgs", &[]), diff --git a/lib/foundation/fabro-api/src/lib.rs b/lib/foundation/fabro-api/src/lib.rs index 96a83eeb5..7fa81a11e 100644 --- a/lib/foundation/fabro-api/src/lib.rs +++ b/lib/foundation/fabro-api/src/lib.rs @@ -38,33 +38,33 @@ pub mod types { BlockedReason, FailureReason, PendingReason, RunControlAction, RunStatus, SuccessReason, }; pub use fabro_types::{ - AgentEventProps, AgentToolsAvailableProps, AskFabro, AuthMethod, AutomationRef, - BilledModelUsage, BilledTokenCounts, BlobHash, CommandTermination, Conclusion, - ContextWindowBreakdownItem, ContextWindowCategory, ContextWindowCountMethod, - ContextWindowSnapshot, ContextWindowStaleness, ContextWindowWarning, CreateVariableRequest, - DiffStats, DiffSummary, DirtyStatus, EventEnvelope, ExecOutputTail, FailureCategory, - FailureDetail, FailureSignature, GitContext, GitRunTarget, - GitRunTarget as AutomationGitWorkflowSource, IdpIdentity, IntegrationConnectionKind, - IntegrationConnectionState, IntegrationConnectionStatus, IntegrationProvider, - IntegrationStatus, InterviewOption, InterviewQuestionRecord, LlmOutputKind, - McpServerDraft as CreateMcpServerRequest, McpServerReplace as ReplaceMcpServerRequest, - McpServerView as McpServer, McpTransportView, Model, ModelControls, ModelCosts, - ModelFeatures, ModelLimits, ModelRef as BillingModelRef, ModelTestMode, PairId, - PairMessageId, PairMessageRecord, PairMessageRequest, PairRecord, PairStartRequest, - PairStatus, PairTarget, PairTranscriptEntry, PairTranscriptResponse, ParallelBranchId, - ParallelBranchResult, PendingInterviewRecord, PermissionLevel, Principal, Provider, - PullRequest, PullRequestCreation, PullRequestCreationId, PullRequestCreationStatus, - PullRequestDetails, PullRequestDetailsStatus, PullRequestDetailsUnavailableReason, - PullRequestLink, PullRequestMeta, PullRequestResponse, QuestionType, RepositoryRef, - ReviewTarget, ReviewTargetKind, Run, RunApproval, RunApprovalState, RunClientProvenance, - RunEvent, RunEventDetailContentKind, RunEventDetailResponse, RunFailure, RunIntent, - RunIntentArgs, RunPairStatusResponse, RunProjection, RunProvenance, RunRunnableSource, - RunSandbox, RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan, - RunSandboxRuntime, RunServerProvenance, RunSessionMetadata, RunSize, RunTarget, - SandboxDetails, SandboxInfo, SandboxListMeta, SandboxListResponse, SandboxProviderKind, - SandboxProviderLookupError, SandboxService, SandboxServiceListResponse, SecretMetadata, - SecretType, ServerSettings, SessionDetail, SessionId, SessionStatus, SessionSummary, - SessionTurn, SkillActivationSource, SkillSummary, StageCompletion, StageContextWindow, + AgentEventProps, AgentToolsAvailableProps, AskFabro, AuthMethod, AutomationRef, BlobHash, + CommandTermination, Conclusion, ContextWindowBreakdownItem, ContextWindowCategory, + ContextWindowCountMethod, ContextWindowSnapshot, ContextWindowStaleness, + ContextWindowWarning, CreateVariableRequest, DiffStats, DiffSummary, DirtyStatus, + EventEnvelope, ExecOutputTail, FailureCategory, FailureDetail, FailureSignature, + GitContext, GitRunTarget, GitRunTarget as AutomationGitWorkflowSource, IdpIdentity, + IntegrationConnectionKind, IntegrationConnectionState, IntegrationConnectionStatus, + IntegrationProvider, IntegrationStatus, InterviewOption, InterviewQuestionRecord, + LlmOutputKind, McpServerDraft as CreateMcpServerRequest, + McpServerReplace as ReplaceMcpServerRequest, McpServerView as McpServer, McpTransportView, + Model, ModelControls, ModelCosts, ModelFeatures, ModelLimits, ModelRef as UsageModelRef, + ModelTestMode, ModelUsage, PairId, PairMessageId, PairMessageRecord, PairMessageRequest, + PairRecord, PairStartRequest, PairStatus, PairTarget, PairTranscriptEntry, + PairTranscriptResponse, ParallelBranchId, ParallelBranchResult, PendingInterviewRecord, + PermissionLevel, Principal, Provider, PullRequest, PullRequestCreation, + PullRequestCreationId, PullRequestCreationStatus, PullRequestDetails, + PullRequestDetailsStatus, PullRequestDetailsUnavailableReason, PullRequestLink, + PullRequestMeta, PullRequestResponse, QuestionType, RepositoryRef, ReviewTarget, + ReviewTargetKind, Run, RunApproval, RunApprovalState, RunClientProvenance, RunEvent, + RunEventDetailContentKind, RunEventDetailResponse, RunFailure, RunIntent, RunIntentArgs, + RunPairStatusResponse, RunProjection, RunProvenance, RunRunnableSource, RunSandbox, + RunSandboxFailure, RunSandboxInstance, RunSandboxKind, RunSandboxPlan, RunSandboxRuntime, + RunServerProvenance, RunSessionMetadata, RunSize, RunTarget, SandboxDetails, SandboxInfo, + SandboxListMeta, SandboxListResponse, SandboxProviderKind, SandboxProviderLookupError, + SandboxService, SandboxServiceListResponse, SecretMetadata, SecretType, ServerSettings, + SessionDetail, SessionId, SessionStatus, SessionSummary, SessionTurn, + SkillActivationSource, SkillSummary, StageCompletion, StageContextWindow, StageContextWindowUnavailableReason, StageHandler, StageId, StageInferenceProjection, StageModelUsage, StageOutcome, StageProjection, StageState, StageToolBatchProjection, SystemActorKind, SystemIntegrationStatus, SystemIntegrationsResponse, TodoListProjection, @@ -74,18 +74,17 @@ pub mod types { }; pub use lithos_llm::catalog::{ModelHandle, ProviderId}; pub use lithos_llm::types::{ - ContentPart, Cost as CompletionCost, CostSource, ErrorKind as LlmErrorKind, Message, - ReasoningEffort, ReasoningOutput, ResponseFormat as CompletionResponseFormat, - RetryClassification as LlmRetryClassification, Role, Speed as BillingSpeed, - TokenCounts as CompletionUsage, ToolChoice as CompletionToolChoice, - ToolDefinition as CompletionToolDefinition, - ToolDefinitionKind as CompletionToolDefinitionKind, + ContentPart, Cost, CostSource, ErrorKind as LlmErrorKind, Message, ReasoningEffort, + ReasoningOutput, ResponseFormat as CompletionResponseFormat, + RetryClassification as LlmRetryClassification, Role, Speed, TokenCounts, + ToolChoice as CompletionToolChoice, ToolDefinition as CompletionToolDefinition, + ToolDefinitionKind as CompletionToolDefinitionKind, Usage, }; /// `StageProjection.agent` is the coding agent's own fold of the stage's /// events; the API reuses pebble's types under the schema names. pub use pebble_coding_agent::events::{ CompactionReason, ErrorData as AgentErrorData, ErrorKind as AgentErrorKind, - FailoverContinuation, FailoverStop, McpToolSummary, TokenUsage, + FailoverContinuation, FailoverStop, McpToolSummary, }; pub use pebble_coding_agent::projection::{ ActivatedSkill as AgentSessionActivatedSkill, diff --git a/lib/foundation/fabro-api/tests/agent_session_projection_round_trip.rs b/lib/foundation/fabro-api/tests/agent_session_projection_round_trip.rs index c5040ccd2..66d6a61b8 100644 --- a/lib/foundation/fabro-api/tests/agent_session_projection_round_trip.rs +++ b/lib/foundation/fabro-api/tests/agent_session_projection_round_trip.rs @@ -30,15 +30,15 @@ use fabro_api::types::{ CompactionReason as ApiCompactionReason, FailoverContinuation as ApiFailoverContinuation, FailoverStop as ApiFailoverStop, LlmErrorKind as ApiLlmErrorKind, LlmRetryClassification as ApiLlmRetryClassification, McpToolSummary as ApiMcpToolSummary, - StageProjection as ApiStageProjection, TokenUsage as ApiTokenUsage, + StageProjection as ApiStageProjection, Usage as ApiUsage, }; use fabro_types::StageProjection; -use lithos_llm::types::{ErrorKind as LlmErrorKind, RetryClassification}; +use lithos_llm::types::{Cost, CostSource, ErrorKind as LlmErrorKind, RetryClassification}; use pebble_coding_agent::events::{ CodingAgentEvent, CodingEvent, CompactionReason, ContextWindowCountMethod, ContextWindowSnapshot, ContextWindowStaleness, ErrorData, ErrorKind, FailoverContinuation, FailoverStop, InputSource, LlmRetryPhase, McpToolSummary, SkillActivationSource, SkillSummary, - TodoCreatedProps, TodoListKind, TodoStatus, TokenUsage, + TodoCreatedProps, TodoListKind, TodoStatus, TokenCounts, Usage, }; use pebble_coding_agent::projection::{ ActivatedSkill, CompactionProjection, DescendantAccount, FailoverStopProjection, @@ -66,7 +66,7 @@ fn agent_session_projection_reuses_pebbles_types() { assert_same_type::(); assert_same_type::(); assert_same_type::(); - assert_same_type::(); + assert_same_type::(); assert_same_type::(); assert_same_type::(); assert_same_type::(); @@ -393,19 +393,14 @@ fn scripted_events() -> Vec { phase: LlmRetryPhase::Open, }), root(CodingEvent::RouteFailover { - from: "openai/gpt-5.2".to_string(), - to: "anthropic/claude-fable-5".to_string(), - attempt: 1, - error: llm_error(), - usage: TokenUsage { - input: 100, - output: 10, - ..TokenUsage::default() - }, - cost_usd_micros: Some(500), - inference_ms: 120, - tool_ms: 30, - continuation: FailoverContinuation::ContinueTurn, + from: "openai/gpt-5.2".to_string(), + to: "anthropic/claude-fable-5".to_string(), + attempt: 1, + error: llm_error(), + usage: priced(100, 10, 500), + inference_ms: 120, + tool_ms: 30, + continuation: FailoverContinuation::ContinueTurn, }), root(CodingEvent::CompactionCompleted { original_turn_count: 20, @@ -413,11 +408,7 @@ fn scripted_events() -> Vec { summary_token_estimate: 500, tracked_file_count: 1, reason: CompactionReason::Threshold, - usage: TokenUsage { - input: 30, - ..TokenUsage::default() - }, - cost_usd_micros: Some(2), + usage: priced(30, 0, 2), }), root(message("anthropic", "claude-fable-5", 50, 5, Some(300))), root(CodingEvent::RouteFailoverStopped { @@ -430,6 +421,21 @@ fn scripted_events() -> Vec { ] } +/// A usage the provider priced. +fn priced(input: u64, output: u64, usd_micros: u64) -> Usage { + Usage { + tokens: TokenCounts { + input, + output, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros, + source: CostSource::Provider, + }), + } +} + fn root(event: CodingEvent) -> CodingAgentEvent { CodingAgentEvent::new("ses_root".to_string(), event, SystemTime::UNIX_EPOCH) } @@ -443,13 +449,17 @@ fn message(provider: &str, model: &str, input: u64, output: u64, cost: Option(); -} - -#[test] -fn billed_token_counts_json_matches_openapi_shape() { - let counts = BilledTokenCounts { - input_tokens: 10, - output_tokens: 20, - total_tokens: 35, - reasoning_tokens: 3, - cache_read_tokens: 1, - cache_write_tokens: 1, - total_usd_micros: Some(42), - }; - - let json = serde_json::to_value(&counts).unwrap(); - assert_eq!(json["input_tokens"], 10); - assert_eq!(json["output_tokens"], 20); - assert_eq!(json["total_tokens"], 35); - assert_eq!(json["reasoning_tokens"], 3); - assert_eq!(json["cache_read_tokens"], 1); - assert_eq!(json["cache_write_tokens"], 1); - assert_eq!(json["total_usd_micros"], 42); - - let round_trip: ApiBilledTokenCounts = serde_json::from_value(json).unwrap(); - assert_eq!(round_trip, counts); -} - -#[test] -fn billed_token_counts_keeps_zero_counts_present() { - let json = serde_json::to_value(BilledTokenCounts::default()).unwrap(); - assert_eq!(json["reasoning_tokens"], 0); - assert_eq!(json["cache_read_tokens"], 0); - assert_eq!(json["cache_write_tokens"], 0); - assert_eq!(json.get("total_usd_micros"), None); - - let round_trip: ApiBilledTokenCounts = serde_json::from_value(json!({ - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 - })) - .unwrap(); - assert_eq!(round_trip, BilledTokenCounts::default()); -} - -fn assert_same_type() { - assert_eq!( - TypeId::of::(), - TypeId::of::(), - "{} should be the same type as {}", - type_name::(), - type_name::() - ); -} diff --git a/lib/foundation/fabro-api/tests/completion_usage_round_trip.rs b/lib/foundation/fabro-api/tests/completion_usage_round_trip.rs deleted file mode 100644 index 03eef28ec..000000000 --- a/lib/foundation/fabro-api/tests/completion_usage_round_trip.rs +++ /dev/null @@ -1,55 +0,0 @@ -use std::any::{TypeId, type_name}; - -use fabro_api::types::CompletionUsage as ApiCompletionUsage; -use lithos_llm::types::TokenCounts; -use serde_json::json; - -#[test] -fn completion_usage_reuses_canonical_type() { - assert_same_type::(); -} - -#[test] -fn completion_usage_json_matches_openapi_shape() { - let usage = TokenCounts { - input: 10, - output: 20, - reasoning: 3, - cache_read: 4, - cache_write: 5, - }; - - let json = serde_json::to_value(usage).unwrap(); - assert_eq!( - json, - json!({ - "input": 10, - "output": 20, - "reasoning": 3, - "cache_read": 4, - "cache_write": 5 - }) - ); - - let round_trip: ApiCompletionUsage = serde_json::from_value(json).unwrap(); - assert_eq!(round_trip, usage); -} - -#[test] -fn completion_usage_missing_buckets_default_to_zero() { - let round_trip: ApiCompletionUsage = serde_json::from_value(json!({"input": 7})).unwrap(); - assert_eq!(round_trip, TokenCounts { - input: 7, - ..TokenCounts::default() - }); -} - -fn assert_same_type() { - assert_eq!( - TypeId::of::(), - TypeId::of::(), - "{} should be the same type as {}", - type_name::(), - type_name::() - ); -} diff --git a/lib/foundation/fabro-api/tests/cost_source_round_trip.rs b/lib/foundation/fabro-api/tests/cost_round_trip.rs similarity index 50% rename from lib/foundation/fabro-api/tests/cost_source_round_trip.rs rename to lib/foundation/fabro-api/tests/cost_round_trip.rs index 484e5847a..051f6a98e 100644 --- a/lib/foundation/fabro-api/tests/cost_source_round_trip.rs +++ b/lib/foundation/fabro-api/tests/cost_round_trip.rs @@ -1,8 +1,8 @@ use std::any::{TypeId, type_name}; -use fabro_api::types::{CompletionCost as ApiCost, CostSource as ApiCostSource}; +use fabro_api::types::{Cost as ApiCost, CostSource as ApiCostSource}; use lithos_llm::types::{Cost, CostSource}; -use serde_json::json; +use serde_json::{Value, json}; #[test] fn cost_types_reuse_lithos_types() { @@ -18,6 +18,7 @@ fn cost_source_json_matches_openapi_shape() { (CostSource::Application, "application"), ] { assert_eq!(serde_json::to_value(source).unwrap(), json!(wire)); + assert_valid("CostSource", &json!(wire)); assert_eq!( serde_json::from_value::(json!(wire)).unwrap(), source @@ -33,6 +34,7 @@ fn cost_json_matches_openapi_shape() { }; let json = serde_json::to_value(cost).unwrap(); assert_eq!(json, json!({"usd_micros": 125000, "source": "provider"})); + assert_valid("Cost", &json); assert_eq!(serde_json::from_value::(json).unwrap(), cost); } @@ -45,3 +47,32 @@ fn assert_same_type() { type_name::() ); } + +#[expect( + clippy::disallowed_methods, + reason = "a synchronous test reads the spec from the repository once" +)] +fn spec() -> Value { + let text = std::fs::read_to_string(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../../docs/public/api-reference/fabro-api.yaml" + )) + .expect("the OpenAPI spec is in the repository"); + let yaml: serde_yaml::Value = serde_yaml::from_str(&text).expect("the spec parses"); + serde_json::to_value(yaml).expect("the spec is JSON-compatible") +} + +fn assert_valid(schema_name: &str, value: &Value) { + let mut root = spec(); + root["$ref"] = json!(format!("#/components/schemas/{schema_name}")); + let validator = jsonschema::validator_for(&root).expect("the spec compiles as a JSON schema"); + let errors: Vec = validator + .iter_errors(value) + .map(|error| format!("{error} at {}", error.instance_path())) + .collect(); + assert!( + errors.is_empty(), + "{schema_name} rejects {value:#}:\n{}", + errors.join("\n") + ); +} diff --git a/lib/foundation/fabro-api/tests/run_billing_stage_round_trip.rs b/lib/foundation/fabro-api/tests/run_billing_stage_round_trip.rs deleted file mode 100644 index c73963932..000000000 --- a/lib/foundation/fabro-api/tests/run_billing_stage_round_trip.rs +++ /dev/null @@ -1,135 +0,0 @@ -use std::any::{TypeId, type_name}; - -use fabro_api::types::{BillingByModel, BillingModelRef, BillingSpeed, RunBillingStage}; -use fabro_types::{ModelRef, StageState}; -use lithos_llm::types::Speed; -use serde_json::json; - -#[test] -fn billing_model_ref_reuses_domain_type() { - assert_same_type::(); - assert_same_type::(); -} - -fn assert_same_type() { - assert_eq!( - TypeId::of::(), - TypeId::of::(), - "{} should be the same type as {}", - type_name::(), - type_name::() - ); -} - -#[test] -fn run_billing_stage_model_accepts_required_null() { - let value = json!({ - "stage": { - "id": "start", - "name": "start" - }, - "model": null, - "billing": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 - }, - "timing": {"wall_time_ms": 0, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0} - }); - - let stage: RunBillingStage = - serde_json::from_value(value).expect("null stage model should deserialize"); - assert!(stage.model.is_none()); - - let encoded = serde_json::to_value(stage).expect("stage should serialize"); - assert!(encoded.get("model").is_some()); - assert!(encoded["model"].is_null()); -} - -#[test] -fn run_billing_stage_round_trips_terminal_row_with_started_at_and_state() { - let value = json!({ - "stage": { - "id": "build", - "name": "build" - }, - "model": { - "provider": "anthropic", - "model_id": "claude-sonnet-4-5", - "speed": "fast" - }, - "billing": { - "input_tokens": 12, - "output_tokens": 34, - "total_tokens": 46, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 - }, - "timing": {"wall_time_ms": 5500, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, - "started_at": "2026-04-29T12:34:56Z", - "state": "succeeded" - }); - - let stage: RunBillingStage = - serde_json::from_value(value.clone()).expect("terminal stage row should deserialize"); - assert!(stage.started_at.is_some()); - assert_eq!(stage.state, Some(StageState::Succeeded)); - assert_eq!(serde_json::to_value(stage).unwrap(), value); -} - -#[test] -fn billing_by_model_round_trips_provider_model_speed_identity() { - let value = json!({ - "model": { - "provider": "anthropic", - "model_id": "claude-opus-4-6", - "speed": "fast" - }, - "stages": 2, - "billing": { - "input_tokens": 12, - "output_tokens": 34, - "total_tokens": 46, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0, - "total_usd_micros": 123 - } - }); - - let row: BillingByModel = - serde_json::from_value(value.clone()).expect("billing model ref should deserialize"); - assert_eq!(serde_json::to_value(row).unwrap(), value); -} - -#[test] -fn run_billing_stage_round_trips_in_flight_row() { - let value = json!({ - "stage": { - "id": "build", - "name": "build" - }, - "model": null, - "billing": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 - }, - "timing": {"wall_time_ms": 1250, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, - "started_at": "2026-04-29T12:34:56Z", - "state": "running" - }); - - let stage: RunBillingStage = - serde_json::from_value(value.clone()).expect("in-flight stage row should deserialize"); - assert!(stage.model.is_none()); - assert_eq!(stage.state, Some(StageState::Running)); - assert_eq!(serde_json::to_value(stage).unwrap(), value); -} diff --git a/lib/foundation/fabro-api/tests/run_failure_round_trip.rs b/lib/foundation/fabro-api/tests/run_failure_round_trip.rs index b3bd27baa..af9033f85 100644 --- a/lib/foundation/fabro-api/tests/run_failure_round_trip.rs +++ b/lib/foundation/fabro-api/tests/run_failure_round_trip.rs @@ -77,7 +77,7 @@ fn conclusion_json_uses_failure_object() { }), final_git_commit_sha: None, stages: Vec::new(), - billing: None, + usage: None, total_retries: 0, diff: Default::default(), }, diff --git a/lib/foundation/fabro-api/tests/run_projection_round_trip.rs b/lib/foundation/fabro-api/tests/run_projection_round_trip.rs index 4f5c589bb..c98535eed 100644 --- a/lib/foundation/fabro-api/tests/run_projection_round_trip.rs +++ b/lib/foundation/fabro-api/tests/run_projection_round_trip.rs @@ -92,12 +92,13 @@ fn run_projection_round_trips_populated_projection() { "parallel_results": null, "output": "done", "usage": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } }, "state": "running" } diff --git a/lib/foundation/fabro-api/tests/run_summary_round_trip.rs b/lib/foundation/fabro-api/tests/run_summary_round_trip.rs index 76280e5a4..dc01c8590 100644 --- a/lib/foundation/fabro-api/tests/run_summary_round_trip.rs +++ b/lib/foundation/fabro-api/tests/run_summary_round_trip.rs @@ -11,9 +11,10 @@ use fabro_types::status::{RunStatus, SuccessReason}; use fabro_types::{ AskFabro, AskFabroUnavailableReason, AutomationRef, DiffSummary, PullRequestLink, RepositoryProvider, RepositoryRef, ResolvedAutomationGitWorkflowSource, Run, RunApproval, - RunApprovalState, RunBillingSummary, RunId, RunLifecycle, RunLinks, RunOrigin, - RunRunnableSource, RunSize, RunTimestamps, RunTiming, WorkflowRef, fixtures, test_support, + RunApprovalState, RunId, RunLifecycle, RunLinks, RunOrigin, RunRunnableSource, RunSize, + RunTimestamps, RunTiming, WorkflowRef, fixtures, test_support, }; +use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use serde_json::json; #[test] @@ -119,9 +120,17 @@ fn run_summary_json_matches_openapi_shape() { completed_at: None, }, timing: Some(RunTiming::new(42_000, 12_000, 30_000)), - billing: Some(RunBillingSummary { - total_usd_micros: Some(123), - }), + usage: Usage { + tokens: TokenCounts { + input: 10, + output: 5, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: 123, + source: CostSource::Catalog, + }), + }, size: RunSize::Xs, ask_fabro: AskFabro { available: false, @@ -217,8 +226,15 @@ fn run_summary_json_matches_openapi_shape() { "tool_time_ms": 30000, "active_time_ms": 42000 }, - "billing": { - "total_usd_micros": 123 + "usage": { + "tokens": { + "input": 10, + "output": 5, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { "usd_micros": 123, "source": "catalog" } }, "size": "XS", "ask_fabro": { @@ -316,7 +332,7 @@ fn run_summary_deserializes_when_optional_fields_are_absent() { assert_eq!(summary.lifecycle.approval, None); assert_eq!(summary.lifecycle.pending_control, None); assert_eq!(summary.timing.map(|t| t.wall_time_ms), None); - assert_eq!(summary.billing, None); + assert_eq!(summary.usage, Usage::default()); assert_eq!(summary.ask_fabro, AskFabro::default()); assert_eq!(summary.superseded_by, None); assert_eq!(summary.retried_from, None); diff --git a/lib/foundation/fabro-api/tests/run_usage_stage_round_trip.rs b/lib/foundation/fabro-api/tests/run_usage_stage_round_trip.rs new file mode 100644 index 000000000..d6ad96482 --- /dev/null +++ b/lib/foundation/fabro-api/tests/run_usage_stage_round_trip.rs @@ -0,0 +1,170 @@ +use std::any::{TypeId, type_name}; + +use fabro_api::types::{RunUsageStage, Speed as ApiSpeed, UsageByModel, UsageModelRef}; +use fabro_types::{ModelRef, StageState}; +use lithos_llm::types::Speed; +use serde_json::{Value, json}; + +#[test] +fn usage_model_ref_reuses_domain_type() { + assert_same_type::(); + assert_same_type::(); +} + +fn assert_same_type() { + assert_eq!( + TypeId::of::(), + TypeId::of::(), + "{} should be the same type as {}", + type_name::(), + type_name::() + ); +} + +fn zero_usage() -> Value { + json!({ + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } + }) +} + +#[test] +fn run_usage_stage_model_accepts_required_null() { + let value = json!({ + "stage": { + "id": "start", + "name": "start" + }, + "model": null, + "usage": zero_usage(), + "timing": {"wall_time_ms": 0, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0} + }); + assert_valid("RunUsageStage", &value); + + let stage: RunUsageStage = + serde_json::from_value(value).expect("null stage model should deserialize"); + assert!(stage.model.is_none()); + + let encoded = serde_json::to_value(stage).expect("stage should serialize"); + assert!(encoded.get("model").is_some()); + assert!(encoded["model"].is_null()); +} + +#[test] +fn run_usage_stage_round_trips_terminal_row_with_started_at_and_state() { + let value = json!({ + "stage": { + "id": "build", + "name": "build" + }, + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-4-5", + "speed": "fast" + }, + "usage": { + "tokens": { + "input": 12, + "output": 34, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { "usd_micros": 123, "source": "catalog" } + }, + "timing": {"wall_time_ms": 5500, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, + "started_at": "2026-04-29T12:34:56Z", + "state": "succeeded" + }); + assert_valid("RunUsageStage", &value); + + let stage: RunUsageStage = + serde_json::from_value(value.clone()).expect("terminal stage row should deserialize"); + assert!(stage.started_at.is_some()); + assert_eq!(stage.state, Some(StageState::Succeeded)); + assert_eq!(stage.usage.cost.map(|cost| cost.usd_micros), Some(123)); + assert_eq!(serde_json::to_value(stage).unwrap(), value); +} + +#[test] +fn usage_by_model_round_trips_provider_model_speed_identity() { + let value = json!({ + "model": { + "provider": "anthropic", + "model_id": "claude-opus-4-6", + "speed": "fast" + }, + "stages": 2, + "usage": { + "tokens": { + "input": 12, + "output": 34, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { "usd_micros": 123, "source": "provider" } + } + }); + assert_valid("UsageByModel", &value); + + let row: UsageByModel = + serde_json::from_value(value.clone()).expect("usage model ref should deserialize"); + assert_eq!(serde_json::to_value(row).unwrap(), value); +} + +#[test] +fn run_usage_stage_round_trips_in_flight_row() { + let value = json!({ + "stage": { + "id": "build", + "name": "build" + }, + "model": null, + "usage": zero_usage(), + "timing": {"wall_time_ms": 1250, "inference_time_ms": 0, "tool_time_ms": 0, "active_time_ms": 0}, + "started_at": "2026-04-29T12:34:56Z", + "state": "running" + }); + assert_valid("RunUsageStage", &value); + + let stage: RunUsageStage = + serde_json::from_value(value.clone()).expect("in-flight stage row should deserialize"); + assert!(stage.model.is_none()); + assert_eq!(stage.state, Some(StageState::Running)); + assert_eq!(serde_json::to_value(stage).unwrap(), value); +} + +#[expect( + clippy::disallowed_methods, + reason = "a synchronous test reads the spec from the repository once" +)] +fn spec() -> Value { + let text = std::fs::read_to_string(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../../docs/public/api-reference/fabro-api.yaml" + )) + .expect("the OpenAPI spec is in the repository"); + let yaml: serde_yaml::Value = serde_yaml::from_str(&text).expect("the spec parses"); + serde_json::to_value(yaml).expect("the spec is JSON-compatible") +} + +fn assert_valid(schema_name: &str, value: &Value) { + let mut root = spec(); + root["$ref"] = json!(format!("#/components/schemas/{schema_name}")); + let validator = jsonschema::validator_for(&root).expect("the spec compiles as a JSON schema"); + let errors: Vec = validator + .iter_errors(value) + .map(|error| format!("{error} at {}", error.instance_path())) + .collect(); + assert!( + errors.is_empty(), + "{schema_name} rejects {value:#}:\n{}", + errors.join("\n") + ); +} diff --git a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs index bd0ff4a1f..80af0dd17 100644 --- a/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs +++ b/lib/foundation/fabro-api/tests/stage_projection_round_trip.rs @@ -2,16 +2,15 @@ use std::any::{TypeId, type_name}; use fabro_api::types::{ AgentToolsAvailableProps as ApiAgentToolsAvailableProps, - BilledModelUsage as ApiBilledModelUsage, ContextWindowBreakdownItem as ApiContextWindowBreakdownItem, ContextWindowCategory as ApiContextWindowCategory, ContextWindowCountMethod as ApiContextWindowCountMethod, ContextWindowSnapshot as ApiContextWindowSnapshot, ContextWindowStaleness as ApiContextWindowStaleness, ContextWindowWarning as ApiContextWindowWarning, LlmOutputKind as ApiLlmOutputKind, - ParallelBranchResult as ApiParallelBranchResult, PermissionLevel as ApiPermissionLevel, - SkillActivationSource as ApiSkillActivationSource, SkillSummary as ApiSkillSummary, - StageContextWindow as ApiStageContextWindow, + ModelUsage as ApiModelUsage, ParallelBranchResult as ApiParallelBranchResult, + PermissionLevel as ApiPermissionLevel, SkillActivationSource as ApiSkillActivationSource, + SkillSummary as ApiSkillSummary, StageContextWindow as ApiStageContextWindow, StageContextWindowUnavailableReason as ApiStageContextWindowUnavailableReason, StageInferenceProjection as ApiStageInferenceProjection, StageProjection as ApiStageProjection, StageToolBatchProjection as ApiStageToolBatchProjection, @@ -19,57 +18,64 @@ use fabro_api::types::{ ToolSource as ApiToolSource, ToolSummary as ApiToolSummary, }; use fabro_types::{ - AgentToolsAvailableProps, BilledModelUsage, ContextWindowBreakdownItem, ContextWindowCategory, + AgentToolsAvailableProps, ContextWindowBreakdownItem, ContextWindowCategory, ContextWindowCountMethod, ContextWindowSnapshot, ContextWindowStaleness, ContextWindowWarning, - LlmOutputKind, ModelRef, ParallelBranchId, ParallelBranchResult, PermissionLevel, + LlmOutputKind, ModelRef, ModelUsage, ParallelBranchId, ParallelBranchResult, PermissionLevel, SkillActivationSource, SkillSummary, StageContextWindow, StageContextWindowUnavailableReason, StageId, StageInferenceProjection, StageProjection, StageToolBatchProjection, TodoListKind, TodoListProjection, ToolCategory, ToolSource, ToolSummary, }; use lithos_llm::catalog::{ModelId, ProviderId}; -use lithos_llm::types::TokenCounts; +use lithos_llm::types::{Cost, CostSource, TokenCounts, Usage}; use serde_json::json; #[test] fn stage_projection_reuses_canonical_type() { assert_same_type::(); - assert_same_type::(); + assert_same_type::(); } #[test] -fn billing_by_model_rows_match_openapi_json_shape() { - let row = BilledModelUsage { - model: ModelRef::new(ProviderId::new("openai"), ModelId::new("gpt-5.4")), - tokens: TokenCounts { - input: 107, - output: 51, - ..TokenCounts::default() +fn usage_by_model_rows_match_openapi_json_shape() { + let row = ModelUsage::new( + ModelRef::new(ProviderId::new("openai"), ModelId::new("gpt-5.4")), + Usage { + tokens: TokenCounts { + input: 107, + output: 51, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: 321, + source: CostSource::Catalog, + }), }, - total_usd_micros: Some(321), - }; + ); let value = serde_json::to_value(&row).unwrap(); assert_eq!( value, json!({ "model": { "provider": "openai", "model_id": "gpt-5.4" }, - "tokens": { - "input": 107, - "output": 51, - "reasoning": 0, - "cache_read": 0, - "cache_write": 0 - }, - "total_usd_micros": 321 + "usage": { + "tokens": { + "input": 107, + "output": 51, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { "usd_micros": 321, "source": "catalog" } + } }) ); - let api_row: ApiBilledModelUsage = serde_json::from_value(value).unwrap(); + let api_row: ApiModelUsage = serde_json::from_value(value).unwrap(); assert_eq!(api_row, row); let mut stage = StageProjection::new(std::num::NonZeroU32::new(1).unwrap()); - stage.billing_by_model = vec![row.clone()]; + stage.usage_by_model = vec![row.clone()]; let stage_json = serde_json::to_value(&stage).unwrap(); assert_eq!( - stage_json["billing_by_model"], + stage_json["usage_by_model"], json!([serde_json::to_value(&row).unwrap()]) ); let without: StageProjection = serde_json::from_value(json!({ @@ -84,21 +90,22 @@ fn billing_by_model_rows_match_openapi_json_shape() { "parallel_results": null, "output": null, "usage": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } }, "state": "running" })) .unwrap(); - assert!(without.billing_by_model.is_empty()); + assert!(without.usage_by_model.is_empty()); assert!( serde_json::to_value(&without) .unwrap() - .get("billing_by_model") + .get("usage_by_model") .is_none(), "no rows, nothing on the wire" ); @@ -207,12 +214,13 @@ fn stage_projection_without_inference_round_trips() { "parallel_results": null, "output": null, "usage": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } }, "state": "running" }); @@ -277,12 +285,13 @@ fn stage_projection_round_trips_representative_json() { "active_time_ms": 0 }, "usage": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - "reasoning_tokens": 0, - "cache_read_tokens": 0, - "cache_write_tokens": 0 + "tokens": { + "input": 0, + "output": 0, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + } }, "permission_level": "read-only", "agent_tools": [ diff --git a/lib/foundation/fabro-api/tests/token_counts_round_trip.rs b/lib/foundation/fabro-api/tests/token_counts_round_trip.rs new file mode 100644 index 000000000..6ca978fad --- /dev/null +++ b/lib/foundation/fabro-api/tests/token_counts_round_trip.rs @@ -0,0 +1,85 @@ +use std::any::{TypeId, type_name}; + +use fabro_api::types::TokenCounts as ApiTokenCounts; +use lithos_llm::types::TokenCounts; +use serde_json::{Value, json}; + +#[test] +fn token_counts_reuses_canonical_type() { + assert_same_type::(); +} + +#[test] +fn token_counts_json_matches_openapi_shape() { + let tokens = TokenCounts { + input: 10, + output: 20, + reasoning: 3, + cache_read: 4, + cache_write: 5, + }; + + let json = serde_json::to_value(tokens).unwrap(); + assert_eq!( + json, + json!({ + "input": 10, + "output": 20, + "reasoning": 3, + "cache_read": 4, + "cache_write": 5 + }) + ); + assert_valid("TokenCounts", &json); + + let round_trip: ApiTokenCounts = serde_json::from_value(json).unwrap(); + assert_eq!(round_trip, tokens); +} + +#[test] +fn token_counts_missing_buckets_default_to_zero() { + let round_trip: ApiTokenCounts = serde_json::from_value(json!({"input": 7})).unwrap(); + assert_eq!(round_trip, TokenCounts { + input: 7, + ..TokenCounts::default() + }); +} + +fn assert_same_type() { + assert_eq!( + TypeId::of::(), + TypeId::of::(), + "{} should be the same type as {}", + type_name::(), + type_name::() + ); +} + +#[expect( + clippy::disallowed_methods, + reason = "a synchronous test reads the spec from the repository once" +)] +fn spec() -> Value { + let text = std::fs::read_to_string(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../../docs/public/api-reference/fabro-api.yaml" + )) + .expect("the OpenAPI spec is in the repository"); + let yaml: serde_yaml::Value = serde_yaml::from_str(&text).expect("the spec parses"); + serde_json::to_value(yaml).expect("the spec is JSON-compatible") +} + +fn assert_valid(schema_name: &str, value: &Value) { + let mut root = spec(); + root["$ref"] = json!(format!("#/components/schemas/{schema_name}")); + let validator = jsonschema::validator_for(&root).expect("the spec compiles as a JSON schema"); + let errors: Vec = validator + .iter_errors(value) + .map(|error| format!("{error} at {}", error.instance_path())) + .collect(); + assert!( + errors.is_empty(), + "{schema_name} rejects {value:#}:\n{}", + errors.join("\n") + ); +} diff --git a/lib/foundation/fabro-api/tests/usage_round_trip.rs b/lib/foundation/fabro-api/tests/usage_round_trip.rs new file mode 100644 index 000000000..bd9b0761b --- /dev/null +++ b/lib/foundation/fabro-api/tests/usage_round_trip.rs @@ -0,0 +1,160 @@ +//! Every usage on the API is lithos-llm's `Usage`: tokens with an optional +//! priced cost. These tests prove the API type is lithos-llm's own, and that +//! the `Usage` and `ModelUsage` schemas describe its serde shape, including +//! the absent `cost`. + +use std::any::{TypeId, type_name}; + +use fabro_api::types::{ModelUsage as ApiModelUsage, Usage as ApiUsage}; +use fabro_types::{ModelRef, ModelUsage}; +use lithos_llm::catalog::{ModelId, ProviderId}; +use lithos_llm::types::{Cost, CostSource, Speed, TokenCounts, Usage}; +use serde_json::{Value, json}; + +#[test] +fn usage_types_reuse_canonical_types() { + assert_same_type::(); + assert_same_type::(); +} + +#[test] +fn usage_json_matches_openapi_shape() { + let usage = Usage { + tokens: TokenCounts { + input: 10, + output: 20, + reasoning: 3, + cache_read: 1, + cache_write: 1, + }, + cost: Some(Cost { + usd_micros: 42, + source: CostSource::Catalog, + }), + }; + + let json = serde_json::to_value(usage).unwrap(); + assert_eq!( + json, + json!({ + "tokens": { + "input": 10, + "output": 20, + "reasoning": 3, + "cache_read": 1, + "cache_write": 1 + }, + "cost": { "usd_micros": 42, "source": "catalog" } + }) + ); + assert_valid("Usage", &json); + + let round_trip: ApiUsage = serde_json::from_value(json).unwrap(); + assert_eq!(round_trip, usage); +} + +#[test] +fn usage_omits_an_absent_cost_and_reads_absent_buckets_as_zero() { + let json = serde_json::to_value(Usage::default()).unwrap(); + assert_eq!(json.get("cost"), None); + assert_eq!(json["tokens"]["reasoning"], 0); + assert_valid("Usage", &json); + + let round_trip: ApiUsage = serde_json::from_value(json!({"tokens": {"input": 7}})).unwrap(); + assert_eq!(round_trip, Usage { + tokens: TokenCounts { + input: 7, + ..TokenCounts::default() + }, + cost: None, + }); +} + +#[test] +fn model_usage_json_matches_openapi_shape() { + let usage = ModelUsage::new( + ModelRef::new( + ProviderId::new("anthropic"), + ModelId::new("claude-sonnet-5"), + ) + .with_speed(Some(Speed::Fast)), + Usage { + tokens: TokenCounts { + input: 100, + output: 20, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: 720_000, + source: CostSource::Provider, + }), + }, + ); + + let json = serde_json::to_value(&usage).unwrap(); + assert_eq!( + json, + json!({ + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-5", + "speed": "fast" + }, + "usage": { + "tokens": { + "input": 100, + "output": 20, + "reasoning": 0, + "cache_read": 0, + "cache_write": 0 + }, + "cost": { "usd_micros": 720000, "source": "provider" } + } + }) + ); + assert_valid("ModelUsage", &json); + + let round_trip: ApiModelUsage = serde_json::from_value(json).unwrap(); + assert_eq!(round_trip, usage); +} + +fn assert_same_type() { + assert_eq!( + TypeId::of::(), + TypeId::of::(), + "{} should be the same type as {}", + type_name::(), + type_name::() + ); +} + +#[expect( + clippy::disallowed_methods, + reason = "a synchronous test reads the spec from the repository once" +)] +fn spec() -> Value { + let text = std::fs::read_to_string(concat!( + env!("CARGO_MANIFEST_DIR"), + "/../../../docs/public/api-reference/fabro-api.yaml" + )) + .expect("the OpenAPI spec is in the repository"); + let yaml: serde_yaml::Value = serde_yaml::from_str(&text).expect("the spec parses"); + serde_json::to_value(yaml).expect("the spec is JSON-compatible") +} + +/// Validates `value` against one component schema, with the whole document +/// as the root so `$ref`s resolve. +fn assert_valid(schema_name: &str, value: &Value) { + let mut root = spec(); + root["$ref"] = json!(format!("#/components/schemas/{schema_name}")); + let validator = jsonschema::validator_for(&root).expect("the spec compiles as a JSON schema"); + let errors: Vec = validator + .iter_errors(value) + .map(|error| format!("{error} at {}", error.instance_path())) + .collect(); + assert!( + errors.is_empty(), + "{schema_name} rejects {value:#}:\n{}", + errors.join("\n") + ); +} diff --git a/lib/foundation/fabro-types/src/billing.rs b/lib/foundation/fabro-types/src/billing.rs deleted file mode 100644 index 2f483f73f..000000000 --- a/lib/foundation/fabro-types/src/billing.rs +++ /dev/null @@ -1,465 +0,0 @@ -//! Billing rollup vocabulary. -//! -//! Per-response token usage and cost come from lithos: [`TokenCounts`] holds -//! the five disjoint buckets and [`CostSource`] says where a cost came from. -//! Fabro sums that usage across responses, stages, and runs. The types here -//! are those sums, plus [`ModelRef`], the identity a billed response is -//! grouped under. - -use lithos_llm::catalog::{ModelHandle, ModelId, ProviderId}; -use lithos_llm::types::{Cost, Speed, TokenCounts}; -use serde::{Deserialize, Serialize}; - -const USD_MICROS_PER_USD_F64: f64 = 1_000_000.0; - -#[allow( - clippy::cast_possible_truncation, - clippy::cast_precision_loss, - reason = "Billing rounds bounded finite floats into i64 counters by design." -)] -fn saturating_rounded_f64_to_i64(value: f64) -> i64 { - if !value.is_finite() { - return if value.is_sign_negative() { - i64::MIN - } else { - i64::MAX - }; - } - - if value <= i64::MIN as f64 { - i64::MIN - } else if value >= i64::MAX as f64 { - i64::MAX - } else { - value as i64 - } -} - -fn saturating_u64_to_i64(value: u64) -> i64 { - i64::try_from(value).unwrap_or(i64::MAX) -} - -fn saturating_i64_to_u64(value: i64) -> u64 { - u64::try_from(value).unwrap_or_default() -} - -/// A USD amount in micros (one millionth of a dollar). -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Default, Serialize, Deserialize)] -pub struct UsdMicros(pub i64); - -impl UsdMicros { - #[must_use] - pub fn from_usd(usd: f64) -> Self { - Self(saturating_rounded_f64_to_i64( - (usd * USD_MICROS_PER_USD_F64).round(), - )) - } - - /// Converts a lithos cost into Fabro's signed micros. - #[must_use] - pub fn from_cost(cost: &Cost) -> Self { - Self(saturating_u64_to_i64(cost.usd_micros)) - } - - /// Folds a cost into a running total that stays `None` until a cost is - /// observed (`None` means "no provider data", not $0). - pub fn accumulate(total: &mut Option, cost: Option) { - if let Some(cost) = cost { - *total.get_or_insert_default() += cost; - } - } -} - -impl std::ops::Add for UsdMicros { - type Output = Self; - - fn add(self, rhs: Self) -> Self::Output { - Self(self.0.saturating_add(rhs.0)) - } -} - -impl std::ops::AddAssign for UsdMicros { - fn add_assign(&mut self, rhs: Self) { - *self = *self + rhs; - } -} - -impl std::iter::Sum for UsdMicros { - fn sum>(iter: I) -> Self { - iter.fold(Self::default(), |acc, value| acc + value) - } -} - -/// Adds `rhs` into `total` bucket by bucket with saturation. -pub fn add_usage(total: &mut TokenCounts, rhs: TokenCounts) { - total.input = total.input.saturating_add(rhs.input); - total.output = total.output.saturating_add(rhs.output); - total.reasoning = total.reasoning.saturating_add(rhs.reasoning); - total.cache_read = total.cache_read.saturating_add(rhs.cache_read); - total.cache_write = total.cache_write.saturating_add(rhs.cache_write); -} - -fn accumulate_optional_usd_micros(total: &mut Option, cost: Option) { - let mut typed_total = (*total).map(UsdMicros); - UsdMicros::accumulate(&mut typed_total, cost.map(UsdMicros)); - *total = typed_total.map(|value| value.0); -} - -/// Provider-qualified model identity a billed response is grouped under. -/// -/// Carries the requested speed tier because providers price tiers -/// differently, so two responses from the same model at different speeds are -/// separate billing rows. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct ModelRef { - pub provider: ProviderId, - pub model_id: ModelId, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub speed: Option, -} - -impl ModelRef { - #[must_use] - pub fn new(provider: ProviderId, model_id: ModelId) -> Self { - Self { - provider, - model_id, - speed: None, - } - } - - #[must_use] - pub fn from_handle(handle: &ModelHandle, speed: Option) -> Self { - Self { - provider: handle.provider().clone(), - model_id: handle.model().clone(), - speed, - } - } - - #[must_use] - pub fn with_speed(mut self, speed: Option) -> Self { - self.speed = speed; - self - } - - #[must_use] - pub fn handle(&self) -> ModelHandle { - ModelHandle::new(self.provider.clone(), self.model_id.clone()) - } - - /// Stable ordering key: provider, then model, then speed label. - #[must_use] - pub fn sort_key(&self) -> (&str, &str, &'static str) { - ( - self.provider.as_str(), - self.model_id.as_str(), - self.speed.map_or("", Speed::as_str), - ) - } -} - -impl std::hash::Hash for ModelRef { - fn hash(&self, state: &mut H) { - self.provider.hash(state); - self.model_id.hash(state); - self.speed.map(Speed::as_str).hash(state); - } -} - -impl std::fmt::Display for ModelRef { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{}/{}", self.provider, self.model_id)?; - if let Some(speed) = self.speed { - write!(f, " ({speed})")?; - } - Ok(()) - } -} - -/// Usage and cost of one billed model response. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct BilledModelUsage { - pub model: ModelRef, - pub tokens: TokenCounts, - /// Cost for `tokens`, when the provider reported one or the catalog could - /// price them. `None` means no cost data, not zero. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub total_usd_micros: Option, -} - -impl BilledModelUsage { - #[must_use] - pub fn new(model: ModelRef, tokens: TokenCounts, cost: Option) -> Self { - Self { - model, - tokens, - total_usd_micros: cost.map(|cost| UsdMicros::from_cost(&cost).0), - } - } - - #[must_use] - pub fn model(&self) -> &ModelRef { - &self.model - } - - #[must_use] - pub fn model_id(&self) -> &str { - self.model.model_id.as_str() - } - - #[must_use] - pub fn tokens(&self) -> TokenCounts { - self.tokens - } - - /// Overrides the billed total with a reported cost; `None` leaves the - /// existing value in place. - #[must_use] - pub fn with_reported_cost(mut self, cost: Option) -> Self { - if let Some(cost) = cost { - self.total_usd_micros = Some(cost.0); - } - self - } -} - -/// Token counts summed across one or more responses, with the summed cost. -/// -/// `total_tokens` is the sum of the five buckets. `total_usd_micros` stays -/// `None` until at least one summed response carried a cost. -#[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)] -pub struct BilledTokenCounts { - pub input_tokens: i64, - pub output_tokens: i64, - pub total_tokens: i64, - #[serde(default)] - pub reasoning_tokens: i64, - #[serde(default)] - pub cache_read_tokens: i64, - #[serde(default)] - pub cache_write_tokens: i64, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub total_usd_micros: Option, -} - -impl BilledTokenCounts { - #[must_use] - pub fn from_token_counts(tokens: TokenCounts, total_usd_micros: Option) -> Self { - Self { - input_tokens: saturating_u64_to_i64(tokens.input), - output_tokens: saturating_u64_to_i64(tokens.output), - total_tokens: saturating_u64_to_i64(tokens.total()), - reasoning_tokens: saturating_u64_to_i64(tokens.reasoning), - cache_read_tokens: saturating_u64_to_i64(tokens.cache_read), - cache_write_tokens: saturating_u64_to_i64(tokens.cache_write), - total_usd_micros, - } - } - - #[must_use] - pub fn from_billed_usage(billed: &[BilledModelUsage]) -> Self { - let mut counts = Self::default(); - for entry in billed { - counts.add_billed_usage(entry); - } - counts - } - - /// Returns the five disjoint per-call token buckets, dropping the derived - /// `total_tokens` sum and the optional `total_usd_micros` cost. - #[must_use] - pub fn token_counts(&self) -> TokenCounts { - TokenCounts { - input: saturating_i64_to_u64(self.input_tokens), - output: saturating_i64_to_u64(self.output_tokens), - reasoning: saturating_i64_to_u64(self.reasoning_tokens), - cache_read: saturating_i64_to_u64(self.cache_read_tokens), - cache_write: saturating_i64_to_u64(self.cache_write_tokens), - } - } - - pub fn add_counts(&mut self, source: &Self) { - self.input_tokens = self.input_tokens.saturating_add(source.input_tokens); - self.output_tokens = self.output_tokens.saturating_add(source.output_tokens); - self.total_tokens = self.total_tokens.saturating_add(source.total_tokens); - self.reasoning_tokens = self - .reasoning_tokens - .saturating_add(source.reasoning_tokens); - self.cache_read_tokens = self - .cache_read_tokens - .saturating_add(source.cache_read_tokens); - self.cache_write_tokens = self - .cache_write_tokens - .saturating_add(source.cache_write_tokens); - accumulate_optional_usd_micros(&mut self.total_usd_micros, source.total_usd_micros); - } - - pub fn add_billed_usage(&mut self, usage: &BilledModelUsage) { - self.add_counts(&Self::from_token_counts( - usage.tokens, - usage.total_usd_micros, - )); - } - - pub fn replace_with_billed_usage(&mut self, usage: &BilledModelUsage) { - *self = Self::from_billed_usage(std::slice::from_ref(usage)); - } - - /// Overrides the billed total with a reported cost; `None` leaves any - /// existing value in place. - #[must_use] - pub fn with_reported_cost(mut self, cost: Option) -> Self { - if let Some(cost) = cost { - self.total_usd_micros = Some(cost.0); - } - self - } - - #[must_use] - pub fn is_zero(&self) -> bool { - self.input_tokens == 0 - && self.output_tokens == 0 - && self.total_tokens == 0 - && self.reasoning_tokens == 0 - && self.cache_read_tokens == 0 - && self.cache_write_tokens == 0 - && self.total_usd_micros.unwrap_or(0) == 0 - } -} - -#[cfg(test)] -mod tests { - use lithos_llm::types::CostSource; - use serde_json::json; - - use super::*; - - fn tokens() -> TokenCounts { - TokenCounts { - input: 100, - output: 20, - reasoning: 5, - cache_read: 7, - cache_write: 3, - } - } - - fn model() -> ModelRef { - ModelRef::new( - ProviderId::new("anthropic"), - ModelId::new("claude-sonnet-5"), - ) - } - - #[test] - fn usd_micros_from_usd_rounds_to_nearest_micro() { - assert_eq!(UsdMicros::from_usd(0.012_345), UsdMicros(12_345)); - assert_eq!(UsdMicros::from_usd(1.0), UsdMicros(1_000_000)); - assert_eq!(UsdMicros::from_usd(f64::INFINITY), UsdMicros(i64::MAX)); - } - - #[test] - fn usd_micros_from_cost_saturates() { - let cost = Cost { - usd_micros: u64::MAX, - source: CostSource::Provider, - }; - assert_eq!(UsdMicros::from_cost(&cost), UsdMicros(i64::MAX)); - } - - #[test] - fn accumulate_stays_none_without_costs() { - let mut total = None; - UsdMicros::accumulate(&mut total, None); - assert_eq!(total, None); - UsdMicros::accumulate(&mut total, Some(UsdMicros(5))); - UsdMicros::accumulate(&mut total, None); - UsdMicros::accumulate(&mut total, Some(UsdMicros(7))); - assert_eq!(total, Some(UsdMicros(12))); - } - - #[test] - fn billed_token_counts_from_token_counts_sums_total() { - let counts = BilledTokenCounts::from_token_counts(tokens(), Some(42)); - assert_eq!(counts.input_tokens, 100); - assert_eq!(counts.output_tokens, 20); - assert_eq!(counts.reasoning_tokens, 5); - assert_eq!(counts.cache_read_tokens, 7); - assert_eq!(counts.cache_write_tokens, 3); - assert_eq!(counts.total_tokens, 135); - assert_eq!(counts.total_usd_micros, Some(42)); - assert_eq!(counts.token_counts(), tokens()); - } - - #[test] - fn billed_token_counts_sum_billed_usage_and_costs() { - let priced = BilledModelUsage::new( - model(), - tokens(), - Some(Cost { - usd_micros: 10, - source: CostSource::Catalog, - }), - ); - let unpriced = BilledModelUsage::new(model(), tokens(), None); - let counts = BilledTokenCounts::from_billed_usage(&[priced, unpriced]); - assert_eq!(counts.input_tokens, 200); - assert_eq!(counts.total_tokens, 270); - assert_eq!(counts.total_usd_micros, Some(10)); - } - - #[test] - fn billed_token_counts_without_costs_report_none() { - let counts = - BilledTokenCounts::from_billed_usage(&[BilledModelUsage::new(model(), tokens(), None)]); - assert_eq!(counts.total_usd_micros, None); - assert!(!counts.is_zero()); - assert!(BilledTokenCounts::default().is_zero()); - } - - #[test] - fn billed_model_usage_serializes_lithos_token_buckets() { - let usage = BilledModelUsage::new(model().with_speed(Some(Speed::Fast)), tokens(), None); - let value = serde_json::to_value(&usage).unwrap(); - assert_eq!( - value, - json!({ - "model": { - "provider": "anthropic", - "model_id": "claude-sonnet-5", - "speed": "fast", - }, - "tokens": { - "input": 100, - "output": 20, - "reasoning": 5, - "cache_read": 7, - "cache_write": 3, - }, - }) - ); - let back: BilledModelUsage = serde_json::from_value(value).unwrap(); - assert_eq!(back, usage); - } - - #[test] - fn model_ref_hash_distinguishes_speed_tiers() { - use std::collections::HashSet; - - let mut set = HashSet::new(); - set.insert(model()); - set.insert(model().with_speed(Some(Speed::Fast))); - set.insert(model().with_speed(Some(Speed::Fast))); - assert_eq!(set.len(), 2); - } - - #[test] - fn model_ref_display_names_the_route_and_speed() { - assert_eq!(model().to_string(), "anthropic/claude-sonnet-5"); - assert_eq!( - model().with_speed(Some(Speed::Fast)).to_string(), - "anthropic/claude-sonnet-5 (fast)" - ); - } -} diff --git a/lib/foundation/fabro-types/src/checkpoint.rs b/lib/foundation/fabro-types/src/checkpoint.rs index 6772034c1..ccd2a63f9 100644 --- a/lib/foundation/fabro-types/src/checkpoint.rs +++ b/lib/foundation/fabro-types/src/checkpoint.rs @@ -4,9 +4,9 @@ use chrono::{DateTime, Utc}; use serde::{Deserialize, Serialize}; use serde_json::Value; -use crate::billing::BilledModelUsage; use crate::failure_signature::FailureSignature; use crate::outcome::Outcome; +use crate::usage::ModelUsage; #[derive(Debug, Clone, Serialize, Deserialize)] pub struct Checkpoint { @@ -16,7 +16,7 @@ pub struct Checkpoint { pub node_retries: HashMap, pub context_values: HashMap, #[serde(default, skip_serializing_if = "HashMap::is_empty")] - pub node_outcomes: HashMap>>, + pub node_outcomes: HashMap>>, #[serde(default, skip_serializing_if = "Option::is_none")] pub next_node_id: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/lib/foundation/fabro-types/src/conclusion.rs b/lib/foundation/fabro-types/src/conclusion.rs index fbc3eb01f..a06697070 100644 --- a/lib/foundation/fabro-types/src/conclusion.rs +++ b/lib/foundation/fabro-types/src/conclusion.rs @@ -1,19 +1,21 @@ use chrono::{DateTime, Utc}; +use lithos_llm::types::Usage; use serde::{Deserialize, Serialize}; use crate::outcome::StageOutcome; -use crate::{BilledTokenCounts, RunDiff, RunFailure, RunTiming, StageTiming}; +use crate::{RunDiff, RunFailure, RunTiming, StageTiming}; #[derive(Debug, Clone, Serialize, Deserialize)] pub struct StageSummary { - pub stage_id: String, - pub stage_label: String, + pub stage_id: String, + pub stage_label: String, /// Per-node timing summed across every visit of the node within this /// conclusion. `wall_time_ms` is the sum of visit wall times. - pub timing: StageTiming, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing_usd_micros: Option, - pub retries: u32, + pub timing: StageTiming, + /// Per-node usage summed across every visit of the node. + #[serde(default)] + pub usage: Usage, + pub retries: u32, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -29,8 +31,10 @@ pub struct Conclusion { pub final_git_commit_sha: Option, #[serde(default, skip_serializing_if = "Vec::is_empty")] pub stages: Vec, + /// The run's usage summed across every stage visit; `None` for a run + /// that made no model calls. #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, + pub usage: Option, #[serde(default)] pub total_retries: u32, #[serde(default)] diff --git a/lib/foundation/fabro-types/src/event_envelope.rs b/lib/foundation/fabro-types/src/event_envelope.rs index 59f1ba21a..e24de806f 100644 --- a/lib/foundation/fabro-types/src/event_envelope.rs +++ b/lib/foundation/fabro-types/src/event_envelope.rs @@ -40,11 +40,10 @@ mod tests { artifact_count: 0, status: "success".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), }; let envelope = EventEnvelope { seq: 7, event }; @@ -84,11 +83,10 @@ mod tests { artifact_count: 1, status: "success".to_string(), reason: SuccessReason::Completed, - total_usd_micros: None, final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }), }; let envelope = EventEnvelope { seq: 99, event }; diff --git a/lib/foundation/fabro-types/src/lib.rs b/lib/foundation/fabro-types/src/lib.rs index bbc11ede5..d23f01c92 100644 --- a/lib/foundation/fabro-types/src/lib.rs +++ b/lib/foundation/fabro-types/src/lib.rs @@ -2,8 +2,6 @@ extern crate self as fabro_types; pub mod artifact; pub mod auth; -pub mod billing; -pub mod billing_rollup; pub mod blob_hash; pub mod blob_ref; pub mod catalog_api; @@ -56,6 +54,8 @@ pub mod system_integrations; pub mod test_support; pub mod timing; pub mod transcript; +pub mod usage; +pub mod usage_rollup; pub mod variable; pub mod workflow_path; pub mod workflow_version; @@ -63,7 +63,6 @@ pub mod workflow_version_id; pub use artifact::ArtifactUpload; pub use auth::{IdpIdentity, IdpIdentityError}; -pub use billing::{BilledModelUsage, BilledTokenCounts, ModelRef, UsdMicros}; pub use blob_hash::BlobHash; pub use blob_ref::{format_blob_ref, parse_blob_ref, parse_managed_blob_file_ref}; pub use catalog_api::{Model, ModelControls, ModelCosts, ModelFeatures, ModelLimits, Provider}; @@ -153,8 +152,8 @@ pub use run_sandbox::{ }; pub use run_summary::{ AskFabro, AskFabroUnavailableReason, AutomationRef, ResolvedAutomationGitWorkflowSource, Run, - RunApproval, RunApprovalState, RunBillingSummary, RunError, RunLifecycle, RunLinks, RunModel, - RunOrigin, RunOriginKind, RunSize, RunTimestamps, WorkflowRef, + RunApproval, RunApprovalState, RunError, RunLifecycle, RunLinks, RunModel, RunOrigin, + RunOriginKind, RunSize, RunTimestamps, WorkflowRef, }; pub use run_title::{ MAX_RUN_TITLE_CHARS, RunTitleError, infer_run_title, normalize_explicit_run_title, @@ -190,6 +189,7 @@ pub use transcript::{ MessageId, MessageKind, MessageSource, PairMessageRef, TranscriptMessage, text_of, tool_call_arguments, tool_result_from_json, tool_result_to_json, }; +pub use usage::{ModelRef, ModelUsage, sum_usage, usage_is_empty}; pub use variable::{ CreateVariableRequest, UpdateVariableRequest, Variable, VariableListResponse, is_env_style_name, }; diff --git a/lib/foundation/fabro-types/src/outcome.rs b/lib/foundation/fabro-types/src/outcome.rs index 02810714a..43f507317 100644 --- a/lib/foundation/fabro-types/src/outcome.rs +++ b/lib/foundation/fabro-types/src/outcome.rs @@ -9,7 +9,7 @@ use serde_json::Value; use strum::{Display, EnumString, IntoStaticStr}; use crate::{ - BilledModelUsage, ExecOutputTail, FailureSignature, OnFailure, ResolvedOnFailure, StageTiming, + ExecOutputTail, FailureSignature, ModelUsage, OnFailure, ResolvedOnFailure, StageTiming, SystemActorKind, }; @@ -275,11 +275,11 @@ pub struct Outcome { pub failure: Option, #[serde(default)] pub usage: M, - /// The stage's billing split by model, for a stage whose agent ran + /// The stage's usage split by model, for a stage whose agent ran /// subagents: the root's route and each subagent's own model. Empty /// otherwise; `usage` is then the one row. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub usage_by_model: Vec, + pub usage_by_model: Vec, #[serde(default, skip_serializing_if = "Vec::is_empty")] pub files_touched: Vec, /// Stage timing breakdown captured by the workflow engine. diff --git a/lib/foundation/fabro-types/src/run_event/agent.rs b/lib/foundation/fabro-types/src/run_event/agent.rs index 50c660e31..4fea34cc1 100644 --- a/lib/foundation/fabro-types/src/run_event/agent.rs +++ b/lib/foundation/fabro-types/src/run_event/agent.rs @@ -240,7 +240,7 @@ pub struct AgentSteerDroppedProps { mod tests { use std::time::{Duration, UNIX_EPOCH}; - use pebble_coding_agent::events::{ErrorData, ErrorKind, FailoverStop, TokenUsage}; + use pebble_coding_agent::events::{ErrorData, ErrorKind, FailoverStop, Usage}; use serde_json::json; use super::*; @@ -302,9 +302,7 @@ mod tests { CodingEvent::AssistantMessage { text: String::new(), model: "gpt-5.4".to_string(), - usage: TokenUsage::default(), - cost_usd_micros: None, - cost_source: None, + usage: Usage::default(), tool_call_count: 0, context_window: None, reasoning: None, diff --git a/lib/foundation/fabro-types/src/run_event/mod.rs b/lib/foundation/fabro-types/src/run_event/mod.rs index 4e036eb11..9bbbb664d 100644 --- a/lib/foundation/fabro-types/src/run_event/mod.rs +++ b/lib/foundation/fabro-types/src/run_event/mod.rs @@ -18,7 +18,7 @@ use serde_json::{Map, Value, json}; pub use session::*; pub use stage::*; -use crate::{BilledTokenCounts, ParallelBranchId, Principal, RunId, StageId}; +use crate::{ParallelBranchId, Principal, RunId, StageId}; /// Maximum accepted body size for `POST /runs/{id}/events`. /// @@ -997,8 +997,8 @@ mod tests { use std::time::{Duration, UNIX_EPOCH}; use pebble_coding_agent::events::{ - CodingAgentEvent, CodingEvent, TodoCreatedProps, TodoListKind, TodoStatus, TokenUsage, - ToolCategory, ToolSource, ToolSummary, + CodingAgentEvent, CodingEvent, Cost, CostSource, TodoCreatedProps, TodoListKind, + TodoStatus, TokenCounts, ToolCategory, ToolSource, ToolSummary, Usage, }; use serde_json::json; @@ -1127,13 +1127,17 @@ mod tests { let body = EventBody::Agent(coding_event("code", 1, CodingEvent::AssistantMessage { text: "ok".to_string(), model: "gpt-5.4".to_string(), - usage: TokenUsage { - input: 10, - output: 5, - ..TokenUsage::default() + usage: Usage { + tokens: TokenCounts { + input: 10, + output: 5, + ..TokenCounts::default() + }, + cost: Some(Cost { + usd_micros: 42, + source: CostSource::Provider, + }), }, - cost_usd_micros: Some(42), - cost_source: None, tool_call_count: 0, context_window: None, reasoning: None, @@ -1141,11 +1145,10 @@ mod tests { let value = serde_json::to_value(&body).unwrap(); assert_eq!( value["properties"]["event"]["AssistantMessage"]["usage"], - json!({"input": 10, "output": 5, "reasoning": 0, "cache_read": 0, "cache_write": 0}) - ); - assert_eq!( - value["properties"]["event"]["AssistantMessage"]["cost_usd_micros"], - 42 + json!({ + "tokens": {"input": 10, "output": 5, "reasoning": 0, "cache_read": 0, "cache_write": 0}, + "cost": {"usd_micros": 42, "source": "provider"} + }) ); } @@ -1190,8 +1193,8 @@ mod tests { status: crate::StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: vec!["next".to_string()], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: Some("done".to_string()), files_touched: vec!["src/main.rs".to_string()], @@ -1626,8 +1629,8 @@ mod tests { status: crate::StageOutcome::Succeeded, preferred_label: None, suggested_next_ids: vec!["next".to_string()], - billing_by_model: Vec::new(), - billing: None, + usage_by_model: Vec::new(), + usage: None, failure: None, notes: Some("done".to_string()), files_touched: vec!["src/main.rs".to_string()], diff --git a/lib/foundation/fabro-types/src/run_event/run.rs b/lib/foundation/fabro-types/src/run_event/run.rs index 33b60f4aa..e23904acc 100644 --- a/lib/foundation/fabro-types/src/run_event/run.rs +++ b/lib/foundation/fabro-types/src/run_event/run.rs @@ -1,8 +1,9 @@ use std::collections::BTreeMap; +use lithos_llm::types::Usage; use serde::{Deserialize, Serialize}; -use super::{BilledTokenCounts, ExecOutputTail, RunNoticeLevel}; +use super::{ExecOutputTail, RunNoticeLevel}; use crate::status::{BlockedReason, PendingReason, SuccessReason}; use crate::{ AutomationRef, BlobHash, DiffSummary, ForkSourceRef, GitContext, Graph, PairId, PairTarget, @@ -243,15 +244,15 @@ pub struct RunCompletedProps { pub status: String, pub reason: SuccessReason, #[serde(default, skip_serializing_if = "Option::is_none")] - pub total_usd_micros: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] pub final_git_commit_sha: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub final_patch: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub diff_summary: Option, + /// The run's usage summed across every stage visit; absent for a run + /// that made no model calls. #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, + pub usage: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -265,8 +266,9 @@ pub struct RunFailedProps { pub final_patch: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub diff_summary: Option, + /// What the run spent before it failed, as on `run.completed`. #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, + pub usage: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] diff --git a/lib/foundation/fabro-types/src/run_event/stage.rs b/lib/foundation/fabro-types/src/run_event/stage.rs index af97b9299..296dc3d41 100644 --- a/lib/foundation/fabro-types/src/run_event/stage.rs +++ b/lib/foundation/fabro-types/src/run_event/stage.rs @@ -5,9 +5,7 @@ use serde::{Deserialize, Serialize}; use serde_json::Value; use super::ExecOutputTail; -use crate::{ - BilledModelUsage, DiffSummary, FailureDetail, Outcome, StageId, StageOutcome, StageTiming, -}; +use crate::{DiffSummary, FailureDetail, ModelUsage, Outcome, StageId, StageOutcome, StageTiming}; #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct StageStartedProps { @@ -36,16 +34,16 @@ pub struct StageCompletedProps { pub preferred_label: Option, #[serde(default, skip_serializing_if = "Vec::is_empty")] pub suggested_next_ids: Vec, - /// The stage's billing: for an agent stage, the whole session tree's + /// The stage's usage: for an agent stage, the whole session tree's /// tokens (the root session and every subagent) under the root's route. #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, - /// `billing` split by model: the root session's route and each - /// subagent's own model, a subagent whose model the catalog does not know - /// billed at the root's. Sums to `billing`. Empty for stages without a - /// coding agent and on events written before it existed. + pub usage: Option, + /// `usage` split by model: the root session's route and each subagent's + /// own model, a subagent whose model the catalog does not know priced at + /// the root's. Sums to `usage`. Empty for stages without a coding agent + /// and on events written before it existed. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub billing_by_model: Vec, + pub usage_by_model: Vec, #[serde(default, skip_serializing_if = "Option::is_none")] pub failure: Option, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -72,20 +70,20 @@ pub struct StageCompletedProps { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct StageFailedProps { - pub index: usize, + pub index: usize, #[serde(default, skip_serializing_if = "Option::is_none")] - pub failure: Option, - pub will_retry: bool, + pub failure: Option, + pub will_retry: bool, /// Per-attempt timing breakdown for this stage visit. #[serde(default)] - pub timing: StageTiming, - /// The stage's billing: for an agent stage that failed after spending, + pub timing: StageTiming, + /// The stage's usage: for an agent stage that failed after spending, /// the whole session tree's tokens under the root's route. #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, - /// `billing` split by model, as on `stage.completed`. + pub usage: Option, + /// `usage` split by model, as on `stage.completed`. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub billing_by_model: Vec, + pub usage_by_model: Vec, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -118,7 +116,7 @@ pub struct PromptCompletedProps { pub model: String, pub provider: String, #[serde(default, skip_serializing_if = "Option::is_none")] - pub billing: Option, + pub usage: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -132,7 +130,7 @@ pub struct CheckpointCompletedProps { #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub context_values: BTreeMap, #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] - pub node_outcomes: BTreeMap>>, + pub node_outcomes: BTreeMap>>, #[serde(default, skip_serializing_if = "Option::is_none")] pub next_node_id: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/lib/foundation/fabro-types/src/run_projection.rs b/lib/foundation/fabro-types/src/run_projection.rs index a50d393c3..f9a1a4a39 100644 --- a/lib/foundation/fabro-types/src/run_projection.rs +++ b/lib/foundation/fabro-types/src/run_projection.rs @@ -3,7 +3,7 @@ use std::collections::{BTreeMap, BTreeSet, HashMap}; use std::num::NonZeroU32; use chrono::{DateTime, Utc}; -use lithos_llm::types::{ReasoningEffort, Speed}; +use lithos_llm::types::{ReasoningEffort, Speed, Usage}; use pebble_coding_agent::events::{ ContextWindowBreakdownItem, ContextWindowCountMethod, ContextWindowSnapshot, ContextWindowStaleness, ContextWindowWarning, LlmOutputKind, PermissionLevel, ToolSummary, @@ -13,11 +13,10 @@ use strum::{Display, EnumString, IntoStaticStr}; use crate::run_event::{AgentSessionActivatedProps, StagePromptProps}; use crate::{ - AgentBackend, BilledModelUsage, BilledTokenCounts, Checkpoint, Conclusion, GitIdentity, - InterviewQuestionRecord, InvalidTransition, ModelRef, ParallelBranchId, PullRequestCreation, - PullRequestLink, RunApproval, RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, - RunTiming, StageCompletion, StageHandler, StageId, StageState, StageTiming, StartRecord, - timing, + AgentBackend, Checkpoint, Conclusion, GitIdentity, InterviewQuestionRecord, InvalidTransition, + ModelRef, ModelUsage, ParallelBranchId, PullRequestCreation, PullRequestLink, RunApproval, + RunControlAction, RunDiff, RunId, RunSandbox, RunSpec, RunStatus, RunTiming, StageCompletion, + StageHandler, StageId, StageState, StageTiming, StartRecord, timing, }; #[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] @@ -293,16 +292,16 @@ pub struct StageProjection { #[serde(default, skip_serializing_if = "Option::is_none")] pub tool_batch: Option, #[serde(default)] - pub usage: BilledTokenCounts, + pub usage: Usage, #[serde(default, skip_serializing_if = "Option::is_none")] pub model: Option, - /// The finished stage's billing split by model, as `stage.completed` or + /// The finished stage's usage split by model, as `stage.completed` or /// `stage.failed` reported it: the root session's route and each /// subagent's own model. Sums to `usage`. Empty while the stage runs and - /// for stages without a coding agent; the billing rollup then bills - /// `usage` to `model`. + /// for stages without a coding agent; the usage rollup then puts + /// `usage` under `model`. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub billing_by_model: Vec, + pub usage_by_model: Vec, #[serde(default, skip_serializing_if = "Option::is_none")] pub permission_level: Option, #[serde(default, skip_serializing_if = "Vec::is_empty")] @@ -411,14 +410,14 @@ impl StageProjection { live_inference_ms: 0, live_tool_ms: 0, tool_batch: None, - usage: BilledTokenCounts::default(), + usage: Usage::default(), model: None, permission_level: None, agent_tools: Vec::new(), inference: None, acp_started_at: None, agent: None, - billing_by_model: Vec::new(), + usage_by_model: Vec::new(), provider_used: None, diff: None, script_invocation: None, @@ -784,7 +783,7 @@ impl RunProjection { /// Whether a graph node is one of the `start`/`exit` boundaries. /// /// Boundary nodes run no work, so callers that report what a run *did* — - /// billing, stage listings, artifact downloads — leave them out. The test + /// usage, stage listings, artifact downloads — leave them out. The test /// is the node's handler type, not its name: a node may be named /// `start` and still do real work. pub fn is_boundary_stage(&self, node_id: &str) -> bool { diff --git a/lib/foundation/fabro-types/src/run_summary.rs b/lib/foundation/fabro-types/src/run_summary.rs index 001b56e3e..a7ae7a7e0 100644 --- a/lib/foundation/fabro-types/src/run_summary.rs +++ b/lib/foundation/fabro-types/src/run_summary.rs @@ -1,6 +1,7 @@ use std::collections::HashMap; use chrono::{DateTime, Utc}; +use lithos_llm::types::{Cost, Usage}; use serde::{Deserialize, Serialize}; use crate::{ @@ -66,8 +67,10 @@ pub struct Run { /// data; populated once a terminal event or partial rollup is available. #[serde(default)] pub timing: Option, + /// The run's usage summed across every stage visit so far: the + /// conclusion's total once the run ended, else the sum of the stages'. #[serde(default)] - pub billing: Option, + pub usage: Usage, #[serde(default)] pub size: RunSize, #[serde(default)] @@ -242,12 +245,6 @@ pub struct RunTimestamps { pub completed_at: Option>, } -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct RunBillingSummary { - #[serde(default)] - pub total_usd_micros: Option, -} - #[derive( Debug, Clone, @@ -279,16 +276,17 @@ impl RunSize { /// Inclusive upper bounds in USD micros for each bucket below [`Self::Xl`], /// ordered smallest to largest. Shared with the SQLite size sort so both /// stay in step. - pub const BUCKET_MAX_USD_MICROS: [(Self, i64); 4] = [ + pub const BUCKET_MAX_USD_MICROS: [(Self, u64); 4] = [ (Self::Xs, 20_000_000), (Self::S, 50_000_000), (Self::M, 100_000_000), (Self::L, 200_000_000), ]; + /// The bucket for a run's cost; a run with no cost data is [`Self::Xs`]. #[must_use] - pub fn from_total_usd_micros(total_usd_micros: Option) -> Self { - let total = total_usd_micros.unwrap_or(0); + pub fn from_cost(cost: Option) -> Self { + let total = cost.map_or(0, |cost| cost.usd_micros); Self::BUCKET_MAX_USD_MICROS .iter() .find(|(_, max)| total <= *max) @@ -298,34 +296,27 @@ impl RunSize { #[cfg(test)] mod tests { + use lithos_llm::types::{Cost, CostSource}; + use super::RunSize; #[test] - fn run_size_uses_billed_usage_thresholds() { - assert_eq!(RunSize::from_total_usd_micros(None), RunSize::Xs); - assert_eq!( - RunSize::from_total_usd_micros(Some(20_000_000)), - RunSize::Xs - ); - assert_eq!(RunSize::from_total_usd_micros(Some(20_000_001)), RunSize::S); - assert_eq!(RunSize::from_total_usd_micros(Some(50_000_000)), RunSize::S); - assert_eq!(RunSize::from_total_usd_micros(Some(50_000_001)), RunSize::M); - assert_eq!( - RunSize::from_total_usd_micros(Some(100_000_000)), - RunSize::M - ); - assert_eq!( - RunSize::from_total_usd_micros(Some(100_000_001)), - RunSize::L - ); - assert_eq!( - RunSize::from_total_usd_micros(Some(200_000_000)), - RunSize::L - ); - assert_eq!( - RunSize::from_total_usd_micros(Some(200_000_001)), - RunSize::Xl - ); + fn run_size_uses_cost_thresholds() { + let cost = |usd_micros: u64| { + Some(Cost { + usd_micros, + source: CostSource::Catalog, + }) + }; + assert_eq!(RunSize::from_cost(None), RunSize::Xs); + assert_eq!(RunSize::from_cost(cost(20_000_000)), RunSize::Xs); + assert_eq!(RunSize::from_cost(cost(20_000_001)), RunSize::S); + assert_eq!(RunSize::from_cost(cost(50_000_000)), RunSize::S); + assert_eq!(RunSize::from_cost(cost(50_000_001)), RunSize::M); + assert_eq!(RunSize::from_cost(cost(100_000_000)), RunSize::M); + assert_eq!(RunSize::from_cost(cost(100_000_001)), RunSize::L); + assert_eq!(RunSize::from_cost(cost(200_000_000)), RunSize::L); + assert_eq!(RunSize::from_cost(cost(200_000_001)), RunSize::Xl); } #[test] diff --git a/lib/foundation/fabro-types/src/timing.rs b/lib/foundation/fabro-types/src/timing.rs index c7a03d1e9..761d49a90 100644 --- a/lib/foundation/fabro-types/src/timing.rs +++ b/lib/foundation/fabro-types/src/timing.rs @@ -153,7 +153,7 @@ impl RunTiming { } /// Sum two run timings field-by-field. Used to accumulate aggregate - /// billing totals across completed runs. + /// usage totals across completed runs. #[must_use] pub fn saturating_add(&self, other: &Self) -> Self { Self::new( diff --git a/lib/foundation/fabro-types/src/transcript.rs b/lib/foundation/fabro-types/src/transcript.rs index f81384e1e..ce278af48 100644 --- a/lib/foundation/fabro-types/src/transcript.rs +++ b/lib/foundation/fabro-types/src/transcript.rs @@ -11,11 +11,11 @@ use lithos_llm::types::{ContentPart, TokenCounts, ToolCall, ToolResult}; use serde::{Deserialize, Serialize}; use strum::{Display, EnumString, IntoStaticStr}; -use crate::billing::ModelRef; use crate::id::ulid_id; use crate::pair::{PairId, PairMessageId}; use crate::principal::Principal; use crate::session::TurnId; +use crate::usage::ModelRef; ulid_id!(MessageId); diff --git a/lib/foundation/fabro-types/src/usage.rs b/lib/foundation/fabro-types/src/usage.rs new file mode 100644 index 000000000..bb86cf521 --- /dev/null +++ b/lib/foundation/fabro-types/src/usage.rs @@ -0,0 +1,267 @@ +//! Usage vocabulary. +//! +//! Token counts and cost come from lithos: [`Usage`] holds the five disjoint +//! token buckets ([`TokenCounts`]) and, when known, what they cost +//! ([`Cost`], with the [`CostSource`] it came from). Fabro sums that usage +//! across responses, stages, and runs with [`Usage::saturating_add`]. The +//! types here are [`ModelRef`], the identity a usage is grouped under, and +//! [`ModelUsage`], a usage with that identity. +//! +//! [`TokenCounts`]: lithos_llm::types::TokenCounts +//! [`Cost`]: lithos_llm::types::Cost +//! [`CostSource`]: lithos_llm::types::CostSource + +use lithos_llm::catalog::{ModelHandle, ModelId, ProviderId}; +use lithos_llm::types::{Speed, Usage}; +use serde::{Deserialize, Serialize}; + +/// Provider-qualified model identity a usage is grouped under. +/// +/// Carries the requested speed tier because providers price tiers +/// differently, so two responses from the same model at different speeds are +/// separate usage rows. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct ModelRef { + pub provider: ProviderId, + pub model_id: ModelId, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub speed: Option, +} + +impl ModelRef { + #[must_use] + pub fn new(provider: ProviderId, model_id: ModelId) -> Self { + Self { + provider, + model_id, + speed: None, + } + } + + #[must_use] + pub fn from_handle(handle: &ModelHandle, speed: Option) -> Self { + Self { + provider: handle.provider().clone(), + model_id: handle.model().clone(), + speed, + } + } + + #[must_use] + pub fn with_speed(mut self, speed: Option) -> Self { + self.speed = speed; + self + } + + #[must_use] + pub fn handle(&self) -> ModelHandle { + ModelHandle::new(self.provider.clone(), self.model_id.clone()) + } + + /// Stable ordering key: provider, then model, then speed label. + #[must_use] + pub fn sort_key(&self) -> (&str, &str, &'static str) { + ( + self.provider.as_str(), + self.model_id.as_str(), + self.speed.map_or("", Speed::as_str), + ) + } +} + +impl std::hash::Hash for ModelRef { + fn hash(&self, state: &mut H) { + self.provider.hash(state); + self.model_id.hash(state); + self.speed.map(Speed::as_str).hash(state); + } +} + +impl std::fmt::Display for ModelRef { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}/{}", self.provider, self.model_id)?; + if let Some(speed) = self.speed { + write!(f, " ({speed})")?; + } + Ok(()) + } +} + +/// Usage grouped under one model: one response, or one model's share of a +/// stage. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct ModelUsage { + pub model: ModelRef, + #[serde(default)] + pub usage: Usage, +} + +impl ModelUsage { + #[must_use] + pub fn new(model: ModelRef, usage: Usage) -> Self { + Self { model, usage } + } + + #[must_use] + pub fn model(&self) -> &ModelRef { + &self.model + } + + #[must_use] + pub fn model_id(&self) -> &str { + self.model.model_id.as_str() + } +} + +/// Sums usages with [`Usage::saturating_add`]; the empty sum is +/// [`Usage::default`]. +pub fn sum_usage(usages: impl IntoIterator) -> Usage { + usages + .into_iter() + .fold(Usage::default(), Usage::saturating_add) +} + +/// No tokens and no cost data: the usage of something that made no model +/// calls. +#[must_use] +pub fn usage_is_empty(usage: &Usage) -> bool { + *usage == Usage::default() +} + +#[cfg(test)] +mod tests { + use lithos_llm::types::{Cost, CostSource, TokenCounts}; + use serde_json::json; + + use super::*; + + fn tokens() -> TokenCounts { + TokenCounts { + input: 100, + output: 20, + reasoning: 5, + cache_read: 7, + cache_write: 3, + } + } + + fn model() -> ModelRef { + ModelRef::new( + ProviderId::new("anthropic"), + ModelId::new("claude-sonnet-5"), + ) + } + + #[test] + fn sum_usage_adds_tokens_and_keeps_a_shared_cost_source() { + let priced = Usage { + tokens: tokens(), + cost: Some(Cost { + usd_micros: 10, + source: CostSource::Catalog, + }), + }; + let total = sum_usage([priced, priced, Usage::default()]); + assert_eq!(total.tokens.input, 200); + assert_eq!(total.tokens.cache_write, 6); + assert_eq!( + total.cost, + Some(Cost { + usd_micros: 20, + source: CostSource::Catalog, + }) + ); + } + + #[test] + fn sum_usage_drops_the_cost_once_an_unpriced_part_used_tokens() { + let priced = Usage { + tokens: tokens(), + cost: Some(Cost { + usd_micros: 10, + source: CostSource::Catalog, + }), + }; + let unpriced = Usage::from(tokens()); + let total = sum_usage([priced, unpriced]); + assert_eq!(total.tokens.input, 200); + assert_eq!(total.cost, None); + assert_eq!(sum_usage([]), Usage::default()); + } + + #[test] + fn usage_is_empty_only_without_tokens_and_cost() { + assert!(usage_is_empty(&Usage::default())); + assert!(!usage_is_empty(&Usage::from(tokens()))); + assert!(!usage_is_empty(&Usage { + tokens: TokenCounts::default(), + cost: Some(Cost { + usd_micros: 1, + source: CostSource::Provider, + }), + })); + } + + #[test] + fn model_usage_serializes_lithos_usage_shape() { + let usage = ModelUsage::new(model().with_speed(Some(Speed::Fast)), Usage { + tokens: tokens(), + cost: Some(Cost { + usd_micros: 42, + source: CostSource::Provider, + }), + }); + let value = serde_json::to_value(&usage).unwrap(); + assert_eq!( + value, + json!({ + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-5", + "speed": "fast", + }, + "usage": { + "tokens": { + "input": 100, + "output": 20, + "reasoning": 5, + "cache_read": 7, + "cache_write": 3, + }, + "cost": { "usd_micros": 42, "source": "provider" }, + }, + }) + ); + let back: ModelUsage = serde_json::from_value(value).unwrap(); + assert_eq!(back, usage); + } + + #[test] + fn model_usage_omits_an_absent_cost() { + let usage = ModelUsage::new(model(), Usage::from(tokens())); + let value = serde_json::to_value(&usage).unwrap(); + assert_eq!(value["usage"].get("cost"), None); + let back: ModelUsage = serde_json::from_value(value).unwrap(); + assert_eq!(back, usage); + } + + #[test] + fn model_ref_hash_distinguishes_speed_tiers() { + use std::collections::HashSet; + + let mut set = HashSet::new(); + set.insert(model()); + set.insert(model().with_speed(Some(Speed::Fast))); + set.insert(model().with_speed(Some(Speed::Fast))); + assert_eq!(set.len(), 2); + } + + #[test] + fn model_ref_display_names_the_route_and_speed() { + assert_eq!(model().to_string(), "anthropic/claude-sonnet-5"); + assert_eq!( + model().with_speed(Some(Speed::Fast)).to_string(), + "anthropic/claude-sonnet-5 (fast)" + ); + } +} diff --git a/lib/foundation/fabro-types/src/billing_rollup.rs b/lib/foundation/fabro-types/src/usage_rollup.rs similarity index 68% rename from lib/foundation/fabro-types/src/billing_rollup.rs rename to lib/foundation/fabro-types/src/usage_rollup.rs index 01b4a31e2..9b236c24f 100644 --- a/lib/foundation/fabro-types/src/billing_rollup.rs +++ b/lib/foundation/fabro-types/src/usage_rollup.rs @@ -1,14 +1,14 @@ use std::collections::HashMap; -use crate::{ - BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageSummary, - StageTiming, -}; +use lithos_llm::types::Usage; + +use crate::usage::usage_is_empty; +use crate::{ModelRef, RunProjection, RunTiming, StageProjection, StageSummary, StageTiming}; #[derive(Debug, Clone, PartialEq)] -pub struct ProjectionBillingStage { +pub struct ProjectionUsageStage { pub node_id: String, - pub billing: BilledTokenCounts, + pub usage: Usage, /// Per-node timing summed across every visit of that node within this /// projection. `wall_time_ms`, `inference_time_ms`, `tool_time_ms`, and /// `active_time_ms` are all summed in lockstep. @@ -17,27 +17,30 @@ pub struct ProjectionBillingStage { } #[derive(Debug, Clone, PartialEq, Eq)] -pub struct ProjectionBillingByModel { - pub model: ModelRef, - pub stages: i64, - pub billing: BilledTokenCounts, +pub struct ProjectionUsageByModel { + pub model: ModelRef, + pub stages: i64, + pub usage: Usage, } #[derive(Debug, Clone, Default, PartialEq)] -pub struct ProjectionBillingRollup { - pub stages: Vec, - pub totals: BilledTokenCounts, - pub by_model: Vec, +pub struct ProjectionUsageRollup { + pub stages: Vec, + pub totals: Usage, + pub by_model: Vec, /// Run-level timing summed across every stage visit. `wall_time_ms` is /// the sum of stage visit wall times (not the run clock duration). - pub timing: RunTiming, - pub billed_visit_count: usize, + pub timing: RunTiming, + /// Stage visits that used tokens or carried a cost. + pub usage_visit_count: usize, } -impl ProjectionBillingRollup { +impl ProjectionUsageRollup { + /// The totals, once at least one stage visit used tokens; `None` for a + /// run that made no model calls. #[must_use] - pub fn billing_if_present(&self) -> Option { - (self.billed_visit_count > 0).then(|| self.totals.clone()) + pub fn usage_if_present(&self) -> Option { + (self.usage_visit_count > 0).then_some(self.totals) } /// Reconstruct the conclusion's per-node summaries from checkpoint and @@ -48,9 +51,9 @@ impl ProjectionBillingRollup { let projection_order = stage_projection_order(projection); // Looping workflows revisit nodes; `completed_nodes` accumulates duplicates // while the other checkpoint maps are keyed by node_id. Dedupe to one row - // per node so the stages table matches the deduped billing total. + // per node so the stages table matches the deduped usage total. if let Some(cp) = projection.current_checkpoint() { - let billing_by_node = self + let usage_by_node = self .stages .iter() .map(|stage| (stage.node_id.as_str(), stage)) @@ -87,13 +90,13 @@ impl ProjectionBillingRollup { .unwrap_or(1) .saturating_sub(1); retries_sum += retries; - let billing = billing_by_node.get(node_id); + let row = usage_by_node.get(node_id); let summary = StageSummary { stage_id: node_id.to_string(), stage_label: node_id.to_string(), - timing: billing.map_or_else(StageTiming::default, |stage| stage.timing), - billing_usd_micros: billing.and_then(|stage| stage.billing.total_usd_micros), + timing: row.map_or_else(StageTiming::default, |stage| stage.timing), + usage: row.map_or_else(Usage::default, |stage| stage.usage), retries, }; stage_rows.push(( @@ -120,29 +123,29 @@ impl ProjectionBillingRollup { } #[must_use] -pub fn billing_rollup_from_projection(projection: &RunProjection) -> ProjectionBillingRollup { +pub fn usage_rollup_from_projection(projection: &RunProjection) -> ProjectionUsageRollup { let mut stage_indices = HashMap::::new(); - let mut stages = Vec::::new(); - let mut by_model = HashMap::::new(); - let mut totals = BilledTokenCounts::default(); + let mut stages = Vec::::new(); + let mut by_model = HashMap::::new(); + let mut totals = Usage::default(); let mut run_timing = RunTiming::default(); - let mut billed_visit_count = 0_usize; + let mut usage_visit_count = 0_usize; for (stage_id, stage) in projection.iter_stages() { if projection.is_boundary_stage(stage_id.node_id()) { continue; } - let usage = &stage.usage; - if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() { + let usage = stage.usage; + if stage.completion.is_none() && stage.timing.is_none() && usage_is_empty(&usage) { continue; } let node_id = stage_id.node_id(); let index = *stage_indices.entry(node_id.to_string()).or_insert_with(|| { let index = stages.len(); - stages.push(ProjectionBillingStage { + stages.push(ProjectionUsageStage { node_id: node_id.to_string(), - billing: BilledTokenCounts::default(), + usage: Usage::default(), timing: StageTiming::default(), model: None, }); @@ -155,28 +158,28 @@ pub fn billing_rollup_from_projection(projection: &RunProjection) -> ProjectionB run_timing = run_timing.saturating_add(&RunTiming::from(timing)); } - if !usage.is_zero() { - billed_visit_count += 1; - row.billing.add_counts(usage); - totals.add_counts(usage); + if !usage_is_empty(&usage) { + usage_visit_count += 1; + row.usage = row.usage.saturating_add(usage); + totals = totals.saturating_add(usage); if let Some(model) = &stage.model { row.model = Some(model.clone()); } - // A completed agent stage says which model billed which tokens: + // A completed agent stage says which model used which tokens: // the root's route and each subagent's own. Until then, and for - // a stage without a coding agent, `usage` bills to `model`. - for (model, billing) in model_rows(stage) { + // a stage without a coding agent, `usage` goes under `model`. + for (model, usage) in model_rows(stage) { let model_entry = by_model .entry(model.clone()) - .or_insert_with(|| ProjectionBillingByModel { + .or_insert_with(|| ProjectionUsageByModel { model, stages: 0, - billing: BilledTokenCounts::default(), + usage: Usage::default(), }); model_entry.stages += 1; - model_entry.billing.add_counts(&billing); + model_entry.usage = model_entry.usage.saturating_add(usage); } } } @@ -184,34 +187,29 @@ pub fn billing_rollup_from_projection(projection: &RunProjection) -> ProjectionB let mut by_model = by_model.into_values().collect::>(); by_model.sort_by(|left, right| left.model.sort_key().cmp(&right.model.sort_key())); - ProjectionBillingRollup { + ProjectionUsageRollup { stages, totals, by_model, timing: run_timing, - billed_visit_count, + usage_visit_count, } } -/// The stage's usage by model: its `billing_by_model` rows when the stage +/// The stage's usage by model: its `usage_by_model` rows when the stage /// completed with them, else its `usage` under its `model`. -fn model_rows(stage: &StageProjection) -> Vec<(ModelRef, BilledTokenCounts)> { - if stage.billing_by_model.is_empty() { +fn model_rows(stage: &StageProjection) -> Vec<(ModelRef, Usage)> { + if stage.usage_by_model.is_empty() { return stage .model .iter() - .map(|model| (model.clone(), stage.usage.clone())) + .map(|model| (model.clone(), stage.usage)) .collect(); } stage - .billing_by_model + .usage_by_model .iter() - .map(|row| { - ( - row.model.clone(), - BilledTokenCounts::from_token_counts(row.tokens, row.total_usd_micros), - ) - }) + .map(|row| (row.model.clone(), row.usage)) .collect() } diff --git a/lib/foundation/fabro-types/tests/run_failure_serde.rs b/lib/foundation/fabro-types/tests/run_failure_serde.rs index ca96879af..100de4200 100644 --- a/lib/foundation/fabro-types/tests/run_failure_serde.rs +++ b/lib/foundation/fabro-types/tests/run_failure_serde.rs @@ -36,7 +36,7 @@ fn run_failed_serializes_nested_failure_contract() { final_git_commit_sha: Some("abc123".to_string()), final_patch: Some("diff --git a/file b/file".to_string()), diff_summary: None, - billing: None, + usage: None, }); let value = serde_json::to_value(&body).expect("run.failed body should serialize"); @@ -90,7 +90,7 @@ fn run_failed_omits_empty_failure_optional_fields() { final_git_commit_sha: None, final_patch: None, diff_summary: None, - billing: None, + usage: None, }); let value = serde_json::to_value(&body).expect("run.failed body should serialize"); @@ -135,7 +135,7 @@ fn conclusion_serializes_rich_failure() { }), final_git_commit_sha: None, stages: Vec::new(), - billing: None, + usage: None, total_retries: 0, diff: RunDiff::default(), }; diff --git a/lib/packages/fabro-api-client/src/api/billing-api.ts b/lib/packages/fabro-api-client/src/api/billing-api.ts deleted file mode 100644 index 93893cf1e..000000000 --- a/lib/packages/fabro-api-client/src/api/billing-api.ts +++ /dev/null @@ -1,122 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -import type { Configuration } from '../configuration'; -import type { AxiosPromise, AxiosInstance, RawAxiosRequestConfig } from 'axios'; -import globalAxios from 'axios'; -// Some imports not used depending on template conditions -// @ts-ignore -import { DUMMY_BASE_URL, assertParamExists, setApiKeyToObject, setBasicAuthToObject, setBearerAuthToObject, setOAuthToObject, setSearchParams, serializeDataIfNeeded, toPathString, createRequestFunction, replaceWithSerializableTypeIfNeeded } from '../common'; -// @ts-ignore -import { BASE_PATH, COLLECTION_FORMATS, type RequestArgs, BaseAPI, RequiredError, operationServerMap } from '../base'; -// @ts-ignore -import type { AggregateBilling } from '../models'; -/** - * BillingApi - axios parameter creator - */ -export const BillingApiAxiosParamCreator = function (configuration?: Configuration) { - return { - /** - * Returns aggregate token counts and billed totals across all completed runs since server start. - * @summary Aggregate Billing - * @param {*} [options] Override http request option. - * @throws {RequiredError} - */ - getAggregateBilling: async (options: RawAxiosRequestConfig = {}): Promise => { - const localVarPath = `/api/v1/billing`; - // use dummy base URL string because the URL constructor only accepts absolute URLs. - const localVarUrlObj = new URL(localVarPath, DUMMY_BASE_URL); - let baseOptions; - if (configuration) { - baseOptions = configuration.baseOptions; - } - - const localVarRequestOptions = { method: 'GET', ...baseOptions, ...options}; - const localVarHeaderParameter = {} as any; - const localVarQueryParameter = {} as any; - - // authentication SessionCookie required - - // authentication BearerAuth required - // http bearer authentication required - await setBearerAuthToObject(localVarHeaderParameter, configuration) - - localVarHeaderParameter['Accept'] = 'application/json'; - - setSearchParams(localVarUrlObj, localVarQueryParameter); - let headersFromBaseOptions = baseOptions && baseOptions.headers ? baseOptions.headers : {}; - localVarRequestOptions.headers = {...localVarHeaderParameter, ...headersFromBaseOptions, ...options.headers}; - - return { - url: toPathString(localVarUrlObj), - options: localVarRequestOptions, - }; - }, - } -}; - -/** - * BillingApi - functional programming interface - */ -export const BillingApiFp = function(configuration?: Configuration) { - const localVarAxiosParamCreator = BillingApiAxiosParamCreator(configuration) - return { - /** - * Returns aggregate token counts and billed totals across all completed runs since server start. - * @summary Aggregate Billing - * @param {*} [options] Override http request option. - * @throws {RequiredError} - */ - async getAggregateBilling(options?: RawAxiosRequestConfig): Promise<(axios?: AxiosInstance, basePath?: string) => AxiosPromise> { - const localVarAxiosArgs = await localVarAxiosParamCreator.getAggregateBilling(options); - const localVarOperationServerIndex = configuration?.serverIndex ?? 0; - const localVarOperationServerBasePath = operationServerMap['BillingApi.getAggregateBilling']?.[localVarOperationServerIndex]?.url; - return (axios, basePath) => createRequestFunction(localVarAxiosArgs, globalAxios, BASE_PATH, configuration)(axios, localVarOperationServerBasePath || basePath); - }, - } -}; - -/** - * BillingApi - factory interface - */ -export const BillingApiFactory = function (configuration?: Configuration, basePath?: string, axios?: AxiosInstance) { - const localVarFp = BillingApiFp(configuration) - return { - /** - * Returns aggregate token counts and billed totals across all completed runs since server start. - * @summary Aggregate Billing - * @param {*} [options] Override http request option. - * @throws {RequiredError} - */ - getAggregateBilling(options?: RawAxiosRequestConfig): AxiosPromise { - return localVarFp.getAggregateBilling(options).then((request) => request(axios, basePath)); - }, - }; -}; - -/** - * BillingApi - object-oriented interface - */ -export class BillingApi extends BaseAPI { - /** - * Returns aggregate token counts and billed totals across all completed runs since server start. - * @summary Aggregate Billing - * @param {*} [options] Override http request option. - * @throws {RequiredError} - */ - public getAggregateBilling(options?: RawAxiosRequestConfig) { - return BillingApiFp(this.configuration).getAggregateBilling(options).then((request) => request(this.axios, this.basePath)); - } -} diff --git a/lib/packages/fabro-api-client/src/models/aggregate-billing-totals.ts b/lib/packages/fabro-api-client/src/models/aggregate-billing-totals.ts deleted file mode 100644 index 7616b4a60..000000000 --- a/lib/packages/fabro-api-client/src/models/aggregate-billing-totals.ts +++ /dev/null @@ -1,60 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { RunTiming } from './run-timing'; - -/** - * Aggregate billing totals across all runs. - */ -export interface AggregateBillingTotals { - /** - * Total number of completed runs. - */ - 'runs': number; - /** - * Total input tokens. - */ - 'input_tokens': number; - /** - * Total output tokens. - */ - 'output_tokens': number; - /** - * Total tokens aggregated across all billing categories. - */ - 'total_tokens': number; - /** - * Total reasoning tokens. - */ - 'reasoning_tokens': number; - /** - * Total cache read tokens. - */ - 'cache_read_tokens': number; - /** - * Total cache write tokens. - */ - 'cache_write_tokens': number; - /** - * Total billed USD amount in micros. - */ - 'total_usd_micros'?: number | null; - /** - * Aggregate timing rollup across every completed run. Active timing sums work across stage visits, so `active_time_ms` can exceed `wall_time_ms`. - */ - 'timing': RunTiming; -} diff --git a/lib/packages/fabro-api-client/src/models/aggregate-billing.ts b/lib/packages/fabro-api-client/src/models/aggregate-billing.ts deleted file mode 100644 index efce18df8..000000000 --- a/lib/packages/fabro-api-client/src/models/aggregate-billing.ts +++ /dev/null @@ -1,32 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { AggregateBillingTotals } from './aggregate-billing-totals'; -// May contain unused imports in some cases -// @ts-ignore -import type { BillingByModel } from './billing-by-model'; - -/** - * Aggregate token counts and billed totals across all runs since server start. - */ -export interface AggregateBilling { - 'totals': AggregateBillingTotals; - /** - * Billing grouped by model. - */ - 'by_model': Array; -} diff --git a/lib/packages/fabro-api-client/src/models/billed-model-usage.ts b/lib/packages/fabro-api-client/src/models/billed-model-usage.ts deleted file mode 100644 index 3d4964992..000000000 --- a/lib/packages/fabro-api-client/src/models/billed-model-usage.ts +++ /dev/null @@ -1,33 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { BillingModelRef } from './billing-model-ref'; -// May contain unused imports in some cases -// @ts-ignore -import type { CompletionUsage } from './completion-usage'; - -/** - * Usage and cost billed to one model: one response, or one model\'s share of a stage. - */ -export interface BilledModelUsage { - 'model': BillingModelRef; - 'tokens': CompletionUsage; - /** - * Cost for `tokens`, when the provider reported one or the catalog could price them. Absent means no cost data, not zero. - */ - 'total_usd_micros'?: number; -} diff --git a/lib/packages/fabro-api-client/src/models/billed-token-counts.ts b/lib/packages/fabro-api-client/src/models/billed-token-counts.ts deleted file mode 100644 index 19413b7c0..000000000 --- a/lib/packages/fabro-api-client/src/models/billed-token-counts.ts +++ /dev/null @@ -1,49 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * Token counts with optional billed USD micros totals. - */ -export interface BilledTokenCounts { - /** - * Number of input tokens consumed. - */ - 'input_tokens': number; - /** - * Number of output tokens generated. - */ - 'output_tokens': number; - /** - * Total billable tokens aggregated across categories. - */ - 'total_tokens': number; - /** - * Number of reasoning tokens. - */ - 'reasoning_tokens': number; - /** - * Number of cache read tokens. - */ - 'cache_read_tokens': number; - /** - * Number of cache write tokens. - */ - 'cache_write_tokens': number; - /** - * Billed USD amount in micros. - */ - 'total_usd_micros'?: number | null; -} diff --git a/lib/packages/fabro-api-client/src/models/billing-by-model.ts b/lib/packages/fabro-api-client/src/models/billing-by-model.ts deleted file mode 100644 index 9c2bf6153..000000000 --- a/lib/packages/fabro-api-client/src/models/billing-by-model.ts +++ /dev/null @@ -1,33 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { BilledTokenCounts } from './billed-token-counts'; -// May contain unused imports in some cases -// @ts-ignore -import type { BillingModelRef } from './billing-model-ref'; - -/** - * Billing statistics grouped by model. - */ -export interface BillingByModel { - 'model': BillingModelRef; - /** - * Number of usage-bearing stage visits that used this model. - */ - 'stages': number; - 'billing': BilledTokenCounts; -} diff --git a/lib/packages/fabro-api-client/src/models/billing-model-ref.ts b/lib/packages/fabro-api-client/src/models/billing-model-ref.ts deleted file mode 100644 index b7f812bb1..000000000 --- a/lib/packages/fabro-api-client/src/models/billing-model-ref.ts +++ /dev/null @@ -1,30 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { BillingSpeed } from './billing-speed'; - -/** - * Provider-qualified billing model identity used for cost estimates. - */ -export interface BillingModelRef { - /** - * LLM provider identifier. - */ - 'provider': string; - 'model_id': string; - 'speed'?: BillingSpeed | null; -} diff --git a/lib/packages/fabro-api-client/src/models/billing-speed.ts b/lib/packages/fabro-api-client/src/models/billing-speed.ts deleted file mode 100644 index 6f3289403..000000000 --- a/lib/packages/fabro-api-client/src/models/billing-speed.ts +++ /dev/null @@ -1,27 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * lithos `Speed`: the requested latency or cost tier. - */ - -export const BillingSpeed = { - FAST: 'fast', - BALANCED: 'balanced', - ECONOMICAL: 'economical' -} as const; - -export type BillingSpeed = typeof BillingSpeed[keyof typeof BillingSpeed]; diff --git a/lib/packages/fabro-api-client/src/models/billing-stage-ref.ts b/lib/packages/fabro-api-client/src/models/billing-stage-ref.ts deleted file mode 100644 index ca95cb907..000000000 --- a/lib/packages/fabro-api-client/src/models/billing-stage-ref.ts +++ /dev/null @@ -1,29 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * Reference to a workflow node in a billing stage row. - */ -export interface BillingStageRef { - /** - * Stage identifier (slug). - */ - 'id': string; - /** - * Human-readable stage name. - */ - 'name': string; -} diff --git a/lib/packages/fabro-api-client/src/models/completion-cost.ts b/lib/packages/fabro-api-client/src/models/completion-cost.ts deleted file mode 100644 index 33775d578..000000000 --- a/lib/packages/fabro-api-client/src/models/completion-cost.ts +++ /dev/null @@ -1,26 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { CostSource } from './cost-source'; - -/** - * lithos `Cost`: a USD amount in micros and where it came from. - */ -export interface CompletionCost { - 'usd_micros': number; - 'source': CostSource; -} diff --git a/lib/packages/fabro-api-client/src/models/completion-usage.ts b/lib/packages/fabro-api-client/src/models/completion-usage.ts deleted file mode 100644 index 11ef8c627..000000000 --- a/lib/packages/fabro-api-client/src/models/completion-usage.ts +++ /dev/null @@ -1,41 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * lithos `TokenCounts`: five disjoint token buckets for one completion. `input` excludes cache reads and writes, while `output` excludes reasoning tokens when the provider reports them separately. - */ -export interface CompletionUsage { - /** - * Uncached prompt tokens. - */ - 'input'?: number; - /** - * Non-reasoning completion tokens. - */ - 'output'?: number; - /** - * Separately reported reasoning tokens. - */ - 'reasoning'?: number; - /** - * Prompt tokens served from a provider cache. - */ - 'cache_read'?: number; - /** - * Prompt tokens written to a provider cache. - */ - 'cache_write'?: number; -} diff --git a/lib/packages/fabro-api-client/src/models/run-billing-stage.ts b/lib/packages/fabro-api-client/src/models/run-billing-stage.ts deleted file mode 100644 index 5db2989dd..000000000 --- a/lib/packages/fabro-api-client/src/models/run-billing-stage.ts +++ /dev/null @@ -1,48 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { BilledTokenCounts } from './billed-token-counts'; -// May contain unused imports in some cases -// @ts-ignore -import type { BillingModelRef } from './billing-model-ref'; -// May contain unused imports in some cases -// @ts-ignore -import type { BillingStageRef } from './billing-stage-ref'; -// May contain unused imports in some cases -// @ts-ignore -import type { StageState } from './stage-state'; -// May contain unused imports in some cases -// @ts-ignore -import type { StageTiming } from './stage-timing'; - -/** - * Token counts and billed totals for one workflow node within a run. Rows are grouped by node; billing and timing sum every visit of that node. - */ -export interface RunBillingStage { - 'stage': BillingStageRef; - 'model': BillingModelRef | null; - 'billing': BilledTokenCounts; - /** - * Per-node timing summed across every visit. `wall_time_ms` is the sum of visit wall times; the active breakdown sums work timing. - */ - 'timing': StageTiming; - /** - * Wall-clock time the latest attempt of this stage started, if known. - */ - 'started_at'?: string | null; - 'state'?: StageState | null; -} diff --git a/lib/packages/fabro-api-client/src/models/run-billing-summary.ts b/lib/packages/fabro-api-client/src/models/run-billing-summary.ts deleted file mode 100644 index c573d568d..000000000 --- a/lib/packages/fabro-api-client/src/models/run-billing-summary.ts +++ /dev/null @@ -1,19 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -export interface RunBillingSummary { - 'total_usd_micros': number | null; -} diff --git a/lib/packages/fabro-api-client/src/models/run-billing-totals.ts b/lib/packages/fabro-api-client/src/models/run-billing-totals.ts deleted file mode 100644 index 66e64f64a..000000000 --- a/lib/packages/fabro-api-client/src/models/run-billing-totals.ts +++ /dev/null @@ -1,56 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { RunTiming } from './run-timing'; - -/** - * Aggregate billing totals across all stages of a run. - */ -export interface RunBillingTotals { - /** - * Run-level timing rollup. `wall_time_ms` is summed across stage visits; active timing sums work across visits. - */ - 'timing': RunTiming; - /** - * Total input tokens consumed. - */ - 'input_tokens': number; - /** - * Total output tokens generated. - */ - 'output_tokens': number; - /** - * Total tokens aggregated across all billing categories. - */ - 'total_tokens': number; - /** - * Total reasoning tokens. - */ - 'reasoning_tokens': number; - /** - * Total cache read tokens. - */ - 'cache_read_tokens': number; - /** - * Total cache write tokens. - */ - 'cache_write_tokens': number; - /** - * Total billed USD amount in micros. - */ - 'total_usd_micros'?: number | null; -} diff --git a/lib/packages/fabro-api-client/src/models/run-billing.ts b/lib/packages/fabro-api-client/src/models/run-billing.ts deleted file mode 100644 index 77cab4145..000000000 --- a/lib/packages/fabro-api-client/src/models/run-billing.ts +++ /dev/null @@ -1,39 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - -// May contain unused imports in some cases -// @ts-ignore -import type { BillingByModel } from './billing-by-model'; -// May contain unused imports in some cases -// @ts-ignore -import type { RunBillingStage } from './run-billing-stage'; -// May contain unused imports in some cases -// @ts-ignore -import type { RunBillingTotals } from './run-billing-totals'; - -/** - * Complete billing breakdown for a single run. - */ -export interface RunBilling { - /** - * Per-node billing breakdown. Each row sums billing and runtime across all visits of that node. - */ - 'stages': Array; - 'totals': RunBillingTotals; - /** - * Billing grouped by model. - */ - 'by_model': Array; -} diff --git a/lib/packages/fabro-api-client/src/models/token-usage.ts b/lib/packages/fabro-api-client/src/models/token-usage.ts deleted file mode 100644 index 1a9946f89..000000000 --- a/lib/packages/fabro-api-client/src/models/token-usage.ts +++ /dev/null @@ -1,41 +0,0 @@ -/* tslint:disable */ -/* eslint-disable */ -/** - * Fabro Run API - * HTTP API for managing Fabro workflow run executions. - * - * The version of the OpenAPI document: 0.2.0 - * - * - * NOTE: This class is auto generated by OpenAPI Generator (https://openapi-generator.tech). - * https://openapi-generator.tech - * Do not edit the class manually. - */ - - - -/** - * Token accounting as the coding agent counts it. The five buckets are disjoint: every token is counted in exactly one, so their plain sum is the total. A bucket that is absent reads as zero. - */ -export interface TokenUsage { - /** - * Prompt tokens that were neither read from nor written to a cache. - */ - 'input'?: number; - /** - * Completion tokens that are not reasoning tokens. - */ - 'output'?: number; - /** - * Completion tokens spent on reasoning, billed at the output rate. - */ - 'reasoning'?: number; - /** - * Prompt tokens served from a provider cache. - */ - 'cache_read'?: number; - /** - * Prompt tokens written into a provider cache. - */ - 'cache_write'?: number; -}