mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-07 03:00:29 +00:00
parent
da475cc865
commit
6e2b9006ea
4 changed files with 487 additions and 122 deletions
493
run.json
493
run.json
|
|
@ -516,7 +516,7 @@
|
|||
"kind": "running"
|
||||
},
|
||||
"status_updated_at": "2026-05-21T22:36:05.837824Z",
|
||||
"last_event_at": "2026-05-21T23:16:18.572394Z",
|
||||
"last_event_at": "2026-05-21T23:16:25.740139Z",
|
||||
"pending_control": null,
|
||||
"checkpoints": [
|
||||
{
|
||||
|
|
@ -1061,9 +1061,9 @@
|
|||
}
|
||||
},
|
||||
{
|
||||
"seq": 0,
|
||||
"seq": 805,
|
||||
"checkpoint": {
|
||||
"timestamp": "2026-05-21T23:16:18.948068Z",
|
||||
"timestamp": "2026-05-21T23:16:25.738527Z",
|
||||
"current_node": "simplify_gpt",
|
||||
"completed_nodes": [
|
||||
"start",
|
||||
|
|
@ -1075,11 +1075,217 @@
|
|||
"simplify_gpt"
|
||||
],
|
||||
"node_retries": {},
|
||||
"context_values": {
|
||||
"outcome": "succeeded",
|
||||
"response.simplify_opus": "All clean. Here's the summary:\n\n## Summary\n\nThree review agents (reuse, quality, efficiency) ran in parallel against the diff. Aggregated findings:\n\n### Fixed\n\n1. **Renamed `BilledTokenCounts::token_counts()` → `to_token_counts()`** (`fabro-model/src/billing.rs`)\n - Quality finding: original name read like a field accessor and hid the fact that the conversion is lossy (drops `total_tokens` and `total_usd_micros`). The `to_` prefix is the established Rust convention for lossy conversions.\n\n2. **Moved pricing logic into `BilledTokenCounts::ensure_priced(catalog, model)`** (`fabro-model/src/billing.rs` + `fabro-workflow/src/billing_rollup.rs`)\n - Quality finding: the previous `stage_usage_with_cost` helper cloned `BilledTokenCounts` across a crate boundary and directly poked the private `total_usd_micros` field, leaking knowledge of how to price a `BilledTokenCounts` out of `fabro-model`. The new method encapsulates that logic next to `add_billed_usage` / `replace_with_billed_usage`, and the rollup-side helper is now a 3-line wrapper.\n\n### Considered and skipped\n\n- **Reuse #1 (refactor `billed_model_usage_from_llm` to use `price_tokens`)** — the \"duplicated\" pricing tail is 3 lines and the surrounding error-vs-option shapes don't compose cleanly. Refactoring adds more indirection than it removes.\n- **`Option<&Catalog>` parameter sprawl** — 6 of 7 call sites pass `None`, but splitting into two functions doubles the public surface for marginal benefit; the single signature is fine.\n- **`pricing_for` memoization across stages** — real but minor (a few `String` clones per stage on a polled endpoint). Worth doing if the path ever shows up in profiles; not justified at the cost of HashMap setup for this fix.\n- **`Cow::Borrowed` for the usage clone** — `BilledTokenCounts` is ~64 bytes of POD; no heap allocation.\n- **Bare `i64` return from `price_tokens`** — matches the established `total_usd_micros: Option<i64>` convention used throughout the codebase.\n\n### Verified\n\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4/4 pass (including the new in-flight pricing test).\n- `cargo nextest run -p fabro-model billed_token_counts` — 5/5 pass.\n- `cargo nextest run -p fabro-server billing` — 7/7 pass.\n- `cargo build --workspace` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.",
|
||||
"failure_class": "",
|
||||
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
||||
"response.simplify_gpt": "Implemented review cleanup on top of the billing cost-on-read changes.\n\nWhat changed:\n- Removed `BilledTokenCounts::ensure_priced`, which was an extra thin helper only used by rollup code.\n- Renamed `to_token_counts()` to `token_counts()` to match the requested API and nearby accessor style.\n- Updated `billing_rollup` pricing helper to avoid cloning stage usage unless a missing cost can actually be priced:\n - borrows existing usage for finalized/already-priced/no-catalog/no-model/unpriceable cases\n - only clones when setting a computed `total_usd_micros`\n\nReview outcome:\n- The parallel review agents did not return actionable findings.\n- I applied the above cleanup for reduced abstraction and less per-read work.\n\nVerification passed:\n- `cargo nextest run -p fabro-workflow billing_rollup`\n- `cargo nextest run -p fabro-server billing`\n- `cargo +nightly-2026-04-14 clippy -q -p fabro-model -p fabro-workflow -p fabro-server --all-targets -- -D warnings`\n\nCurrent working tree has modifications in:\n- `lib/crates/fabro-model/src/billing.rs`\n- `lib/crates/fabro-workflow/src/billing_rollup.rs`",
|
||||
"internal.retry_count.toolchain": 0,
|
||||
"internal.thread_id": "simplify_opus",
|
||||
"failure_signature": "",
|
||||
"internal.retry_count.preflight_compile": 0,
|
||||
"internal.retry_count.start": 0,
|
||||
"current_node": "simplify_gpt",
|
||||
"internal.retry_count.implement": 0,
|
||||
"internal.retry_count.simplify_gpt": 0,
|
||||
"last_response": "Implemented review cleanup on top of the billing cost-on-read changes.\n\nWhat changed:\n- Removed `BilledTokenCounts::ensure_priced`, which was an extra thin helper only used by rollup code.\n- Renamed `",
|
||||
"internal.node_visit_count": 1,
|
||||
"internal.run_id": "01KS6AW28FZVV4M2EHBJA7JMNP",
|
||||
"last_stage": "simplify_gpt",
|
||||
"internal.retry_count.preflight_lint": 0,
|
||||
"thread.start.current_node": "toolchain",
|
||||
"thread.toolchain.current_node": "preflight_compile",
|
||||
"internal.work_dir": "/home/daytona/workspace/fabro",
|
||||
"thread.simplify_opus.current_node": "simplify_gpt",
|
||||
"graph.rankdir": "LR",
|
||||
"internal.retry_count.simplify_opus": 0,
|
||||
"thread.implement.current_node": "simplify_opus",
|
||||
"thread.preflight_lint.current_node": "implement",
|
||||
"thread.preflight_compile.current_node": "preflight_lint",
|
||||
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
||||
"response.implement": "All changes implemented as specified in the plan:\n\n**1. `lib/crates/fabro-model/src/billing.rs`** ✅\n- Added `BilledTokenCounts::token_counts(&self) -> TokenCounts` (drops `total_tokens`/`total_usd_micros`).\n- Added `Catalog::price_tokens(&self, model, tokens) -> Option<i64>` that mirrors the cost lines of `billed_model_usage_from_llm`.\n\n**2. `lib/crates/fabro-workflow/src/billing_rollup.rs`** ✅\n- Changed `billing_rollup_from_projection` signature to take `catalog: Option<&Catalog>`.\n- Added private helper `stage_usage_with_cost` that clones `stage.usage` and prices it when `total_usd_micros.is_none()` and both `catalog` and `stage.model` are available.\n- Uses `priced` in place of `&stage.usage` for the `is_zero` check, `row.billing.add_counts`, `totals.add_counts`, and `model_entry.billing.add_counts`.\n- Updated existing tests to pass `None`; added a new `rollup_prices_in_flight_stage_usage_using_catalog` test (in-flight stage with no completion, non-zero usage, builtin model → `Some(..)` cost on stage row, totals, and `by_model`).\n\n**3. Call sites** ✅\n- `fabro-server/src/server/handler/billing.rs:82` — binds `let catalog = state.catalog();` and passes `Some(&catalog)`.\n- `fabro-server/src/server.rs` (2 aggregate-billing sites) — pass `None`.\n- `fabro-workflow/src/pipeline/finalize.rs` (4 sites) — pass `None`.\n\n**Verification** ✅\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4 tests pass (new + 3 updated).\n- `cargo nextest run -p fabro-server billing` — 7 tests pass.\n- `cargo nextest run -p fabro-model billing` — 23 tests pass.\n- `cargo check --workspace --all-targets` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.\n- Broader test sweep on `fabro-workflow` + `fabro-server`: 1674 passing; the 2 unrelated `*graph*svg*` failures are pre-existing graphviz subprocess env issues (confirmed by stashing my changes and reproducing).",
|
||||
"internal.fidelity": "compact",
|
||||
"graph.goal": "# Plan: Compute LLM cost on-read for in-flight stages\n\n## Context\n\nOn the run billing page (`/runs/{id}/billing`), an active stage shows token\nusage but no dollar cost — cost renders as `—` until the stage completes.\n\nRoot cause: while a stage runs, `AgentMessage` events carry usage built by\n`billed_token_counts_from_llm` (`fabro-workflow/src/outcome.rs:43`), which\nhard-codes `total_usd_micros: None`. Dollar cost is only computed by\n`billed_model_usage_from_llm` (`outcome.rs:14`) — which needs the pricing\n`Catalog` — and that runs only on `StageCompleted`/`PromptCompleted`/`StageFailed`.\nSo an in-flight stage's `StageProjection.usage.total_usd_micros` stays `None`.\n\nFix: price stages whose cost is `None` when the billing rollup is built for a\nread request, using the model + token counts already in the projection. The\nwire contract is unchanged (`total_usd_micros` is already nullable everywhere)\nand the frontend already renders whatever value comes back — no UI change.\n\n## Decisions\n\n- **Price any stage with `total_usd_micros == None`**, not just in-flight ones.\n Completed stages with unpriceable providers (`BillingPolicy::None`) return\n `None` again — harmless; no need to thread `StageState`.\n- **No \"estimated\" label.** Cost-so-far is exact for tokens consumed so far,\n matching the already-unlabeled live token counts and ticking runtime.\n- **Aggregate billing stays finalized-only.** The `BillingAccumulator` call\n sites pass `None` so a run's running estimate is never folded into org-wide\n totals (avoids double-count when the run later finalizes).\n- Per-stage rows, `totals`, and `by_model` are all priced from the same source\n so the billing page stays internally consistent.\n\n## Changes\n\n### 1. `lib/crates/fabro-model/src/billing.rs`\n\n- Add `BilledTokenCounts::token_counts(&self) -> TokenCounts` — drops\n `total_tokens`/`total_usd_micros`, keeps the five disjoint buckets.\n- Add `Catalog::price_tokens(&self, model: &ModelRef, tokens: &TokenCounts) -> Option<i64>`\n next to `pricing_for`/`billing_facts_for`. Body mirrors the cost lines of\n `billed_model_usage_from_llm`: build `ModelBillingFacts` via\n `billing_facts_for`, assemble `ModelBillingInput { ModelUsage { model, tokens }, facts }`,\n then `pricing_for(model).and_then(|p| p.bill(&input)).map(|a| a.0)`. Returns\n `None` when the provider has no billing policy.\n\n### 2. `lib/crates/fabro-workflow/src/billing_rollup.rs`\n\n- Change signature to\n `billing_rollup_from_projection(projection: &RunProjection, catalog: Option<&Catalog>)`.\n- Add a module-private helper `stage_usage_with_cost(catalog, stage) -> BilledTokenCounts`:\n clone `stage.usage`; if `total_usd_micros.is_none()` and both `catalog` and\n `stage.model` are present, set it via `catalog.price_tokens(model, &usage.token_counts())`.\n- In the loop, compute `priced` once per stage and use it in place of\n `&stage.usage` for the `is_zero` check, `row.billing.add_counts`,\n `totals.add_counts`, and `model_entry.billing.add_counts`.\n- Update the existing tests to pass `None`; add one new test: an in-flight\n stage (no `completion`, non-zero `usage` with `total_usd_micros: None`, a\n builtin `model`) yields `Some(..)` cost on the stage row and in `totals` when\n called with `Some(Catalog::builtin())`.\n\n### 3. Call sites of `billing_rollup_from_projection`\n\n- `lib/crates/fabro-server/src/server/handler/billing.rs:82` — bind\n `let catalog = state.catalog();` (returns `Arc<Catalog>`) and pass\n `Some(&catalog)`.\n- `lib/crates/fabro-server/src/server.rs` (2 aggregate-billing sites) — pass `None`.\n- `lib/crates/fabro-workflow/src/pipeline/finalize.rs` (4 sites) — pass `None`\n (stages already priced at completion; pricing would be a no-op anyway).\n\nNo changes to `fabro-api.yaml`, the generated clients, or `apps/fabro-web`.\n\n## Out of scope / known limitation\n\nIn-flight **prompt** stages have no `model` until `PromptCompleted` (only\n`AgentMessage` sets `stage.model` mid-run), so they still show `—` while\nrunning. Acceptable: prompt stages are a single short LLM call. The bug report\nconcerns agent stages, where `model` is available.\n\n## Verification\n\n- `cargo nextest run -p fabro-workflow billing_rollup` — new + updated unit tests pass.\n- `cargo nextest run -p fabro-server billing` — handler conformance still passes.\n- `cargo build --workspace` and `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`.\n- Manual: `fabro server start` + `cd apps/fabro-web && bun run dev`, start a\n workflow with an agent stage, open `/runs/{id}/billing` mid-run — the active\n stage row and totals show a non-`—` dollar amount that grows with tokens.\n"
|
||||
},
|
||||
"node_outcomes": {
|
||||
"simplify_opus": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"response.simplify_opus": "All clean. Here's the summary:\n\n## Summary\n\nThree review agents (reuse, quality, efficiency) ran in parallel against the diff. Aggregated findings:\n\n### Fixed\n\n1. **Renamed `BilledTokenCounts::token_counts()` → `to_token_counts()`** (`fabro-model/src/billing.rs`)\n - Quality finding: original name read like a field accessor and hid the fact that the conversion is lossy (drops `total_tokens` and `total_usd_micros`). The `to_` prefix is the established Rust convention for lossy conversions.\n\n2. **Moved pricing logic into `BilledTokenCounts::ensure_priced(catalog, model)`** (`fabro-model/src/billing.rs` + `fabro-workflow/src/billing_rollup.rs`)\n - Quality finding: the previous `stage_usage_with_cost` helper cloned `BilledTokenCounts` across a crate boundary and directly poked the private `total_usd_micros` field, leaking knowledge of how to price a `BilledTokenCounts` out of `fabro-model`. The new method encapsulates that logic next to `add_billed_usage` / `replace_with_billed_usage`, and the rollup-side helper is now a 3-line wrapper.\n\n### Considered and skipped\n\n- **Reuse #1 (refactor `billed_model_usage_from_llm` to use `price_tokens`)** — the \"duplicated\" pricing tail is 3 lines and the surrounding error-vs-option shapes don't compose cleanly. Refactoring adds more indirection than it removes.\n- **`Option<&Catalog>` parameter sprawl** — 6 of 7 call sites pass `None`, but splitting into two functions doubles the public surface for marginal benefit; the single signature is fine.\n- **`pricing_for` memoization across stages** — real but minor (a few `String` clones per stage on a polled endpoint). Worth doing if the path ever shows up in profiles; not justified at the cost of HashMap setup for this fix.\n- **`Cow::Borrowed` for the usage clone** — `BilledTokenCounts` is ~64 bytes of POD; no heap allocation.\n- **Bare `i64` return from `price_tokens`** — matches the established `total_usd_micros: Option<i64>` convention used throughout the codebase.\n\n### Verified\n\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4/4 pass (including the new in-flight pricing test).\n- `cargo nextest run -p fabro-model billed_token_counts` — 5/5 pass.\n- `cargo nextest run -p fabro-server billing` — 7/7 pass.\n- `cargo build --workspace` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.",
|
||||
"last_stage": "simplify_opus",
|
||||
"last_response": "All clean. Here's the summary:\n\n## Summary\n\nThree review agents (reuse, quality, efficiency) ran in parallel against the diff. Aggregated findings:\n\n### Fixed\n\n1. **Renamed `BilledTokenCounts::token_c"
|
||||
},
|
||||
"notes": "Stage completed: simplify_opus",
|
||||
"usage": {
|
||||
"input": {
|
||||
"usage": {
|
||||
"model": {
|
||||
"provider": "anthropic",
|
||||
"model_id": "claude-opus-4-7"
|
||||
},
|
||||
"tokens": {
|
||||
"input_tokens": 55570,
|
||||
"output_tokens": 17369,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 1126033,
|
||||
"cache_write_tokens": 66184
|
||||
}
|
||||
},
|
||||
"facts": {
|
||||
"algorithm": "anthropic",
|
||||
"cache_write_5m_tokens": 66184,
|
||||
"cache_write_1h_tokens": 0
|
||||
}
|
||||
},
|
||||
"total_usd_micros": 1688741
|
||||
},
|
||||
"files_touched": [
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-model/src/billing.rs",
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-workflow/src/billing_rollup.rs"
|
||||
]
|
||||
},
|
||||
"toolchain": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
||||
},
|
||||
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"usage": null
|
||||
},
|
||||
"preflight_compile": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
||||
},
|
||||
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
||||
"usage": null
|
||||
},
|
||||
"preflight_lint": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
||||
},
|
||||
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
||||
"usage": null
|
||||
},
|
||||
"implement": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"response.implement": "All changes implemented as specified in the plan:\n\n**1. `lib/crates/fabro-model/src/billing.rs`** ✅\n- Added `BilledTokenCounts::token_counts(&self) -> TokenCounts` (drops `total_tokens`/`total_usd_micros`).\n- Added `Catalog::price_tokens(&self, model, tokens) -> Option<i64>` that mirrors the cost lines of `billed_model_usage_from_llm`.\n\n**2. `lib/crates/fabro-workflow/src/billing_rollup.rs`** ✅\n- Changed `billing_rollup_from_projection` signature to take `catalog: Option<&Catalog>`.\n- Added private helper `stage_usage_with_cost` that clones `stage.usage` and prices it when `total_usd_micros.is_none()` and both `catalog` and `stage.model` are available.\n- Uses `priced` in place of `&stage.usage` for the `is_zero` check, `row.billing.add_counts`, `totals.add_counts`, and `model_entry.billing.add_counts`.\n- Updated existing tests to pass `None`; added a new `rollup_prices_in_flight_stage_usage_using_catalog` test (in-flight stage with no completion, non-zero usage, builtin model → `Some(..)` cost on stage row, totals, and `by_model`).\n\n**3. Call sites** ✅\n- `fabro-server/src/server/handler/billing.rs:82` — binds `let catalog = state.catalog();` and passes `Some(&catalog)`.\n- `fabro-server/src/server.rs` (2 aggregate-billing sites) — pass `None`.\n- `fabro-workflow/src/pipeline/finalize.rs` (4 sites) — pass `None`.\n\n**Verification** ✅\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4 tests pass (new + 3 updated).\n- `cargo nextest run -p fabro-server billing` — 7 tests pass.\n- `cargo nextest run -p fabro-model billing` — 23 tests pass.\n- `cargo check --workspace --all-targets` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.\n- Broader test sweep on `fabro-workflow` + `fabro-server`: 1674 passing; the 2 unrelated `*graph*svg*` failures are pre-existing graphviz subprocess env issues (confirmed by stashing my changes and reproducing).",
|
||||
"last_response": "All changes implemented as specified in the plan:\n\n**1. `lib/crates/fabro-model/src/billing.rs`** ✅\n- Added `BilledTokenCounts::token_counts(&self) -> TokenCounts` (drops `total_tokens`/`total_usd_m",
|
||||
"last_stage": "implement"
|
||||
},
|
||||
"notes": "Stage completed: implement",
|
||||
"usage": {
|
||||
"input": {
|
||||
"usage": {
|
||||
"model": {
|
||||
"provider": "anthropic",
|
||||
"model_id": "claude-opus-4-7"
|
||||
},
|
||||
"tokens": {
|
||||
"input_tokens": 92896,
|
||||
"output_tokens": 16017,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 4690781,
|
||||
"cache_write_tokens": 101072
|
||||
}
|
||||
},
|
||||
"facts": {
|
||||
"algorithm": "anthropic",
|
||||
"cache_write_5m_tokens": 101072,
|
||||
"cache_write_1h_tokens": 0
|
||||
}
|
||||
},
|
||||
"total_usd_micros": 3841995
|
||||
},
|
||||
"files_touched": [
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-model/src/billing.rs",
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-server/src/server.rs",
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-server/src/server/handler/billing.rs",
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-workflow/src/billing_rollup.rs",
|
||||
"/home/daytona/workspace/fabro/lib/crates/fabro-workflow/src/pipeline/finalize.rs"
|
||||
]
|
||||
},
|
||||
"simplify_gpt": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"last_response": "Implemented review cleanup on top of the billing cost-on-read changes.\n\nWhat changed:\n- Removed `BilledTokenCounts::ensure_priced`, which was an extra thin helper only used by rollup code.\n- Renamed `",
|
||||
"response.simplify_gpt": "Implemented review cleanup on top of the billing cost-on-read changes.\n\nWhat changed:\n- Removed `BilledTokenCounts::ensure_priced`, which was an extra thin helper only used by rollup code.\n- Renamed `to_token_counts()` to `token_counts()` to match the requested API and nearby accessor style.\n- Updated `billing_rollup` pricing helper to avoid cloning stage usage unless a missing cost can actually be priced:\n - borrows existing usage for finalized/already-priced/no-catalog/no-model/unpriceable cases\n - only clones when setting a computed `total_usd_micros`\n\nReview outcome:\n- The parallel review agents did not return actionable findings.\n- I applied the above cleanup for reduced abstraction and less per-read work.\n\nVerification passed:\n- `cargo nextest run -p fabro-workflow billing_rollup`\n- `cargo nextest run -p fabro-server billing`\n- `cargo +nightly-2026-04-14 clippy -q -p fabro-model -p fabro-workflow -p fabro-server --all-targets -- -D warnings`\n\nCurrent working tree has modifications in:\n- `lib/crates/fabro-model/src/billing.rs`\n- `lib/crates/fabro-workflow/src/billing_rollup.rs`",
|
||||
"last_stage": "simplify_gpt"
|
||||
},
|
||||
"notes": "Stage completed: simplify_gpt",
|
||||
"usage": {
|
||||
"input": {
|
||||
"usage": {
|
||||
"model": {
|
||||
"provider": "openai",
|
||||
"model_id": "gpt-5.5"
|
||||
},
|
||||
"tokens": {
|
||||
"input_tokens": 89620,
|
||||
"output_tokens": 7291,
|
||||
"reasoning_tokens": 3776,
|
||||
"cache_read_tokens": 1726976,
|
||||
"cache_write_tokens": 0
|
||||
}
|
||||
},
|
||||
"facts": {
|
||||
"algorithm": "openai"
|
||||
}
|
||||
},
|
||||
"total_usd_micros": 1643598
|
||||
}
|
||||
},
|
||||
"start": {
|
||||
"status": "succeeded",
|
||||
"usage": null
|
||||
}
|
||||
},
|
||||
"next_node_id": "verify",
|
||||
"git_commit_sha": "208af43dac0b21c58b1b2b68c34c57fb1bd42714",
|
||||
"node_visits": {
|
||||
"toolchain": 1,
|
||||
"preflight_compile": 1,
|
||||
"implement": 1,
|
||||
"start": 1,
|
||||
"simplify_opus": 1,
|
||||
"preflight_lint": 1,
|
||||
"simplify_gpt": 1
|
||||
}
|
||||
},
|
||||
"diff": {
|
||||
"patch": "diff --git a/lib/crates/fabro-model/src/billing.rs b/lib/crates/fabro-model/src/billing.rs\nindex a3be9f470..f06cad06a 100644\n--- a/lib/crates/fabro-model/src/billing.rs\n+++ b/lib/crates/fabro-model/src/billing.rs\n@@ -361,7 +361,7 @@ impl BilledTokenCounts {\n /// Returns the five disjoint per-call token buckets, dropping the derived\n /// `total_tokens` sum and the optional `total_usd_micros` cost.\n #[must_use]\n- pub fn to_token_counts(&self) -> TokenCounts {\n+ pub fn token_counts(&self) -> TokenCounts {\n TokenCounts {\n input_tokens: self.input_tokens,\n output_tokens: self.output_tokens,\n@@ -371,17 +371,6 @@ impl BilledTokenCounts {\n }\n }\n \n- /// Fills in `total_usd_micros` from catalog pricing when it is missing.\n- ///\n- /// No-op when the cost is already known. Used by read-side rollups so\n- /// in-flight stages can show an exact cost for the tokens consumed so far\n- /// without mutating the underlying projection.\n- pub fn ensure_priced(&mut self, catalog: &Catalog, model: &ModelRef) {\n- if self.total_usd_micros.is_none() {\n- self.total_usd_micros = catalog.price_tokens(model, &self.to_token_counts());\n- }\n- }\n-\n pub fn add_counts(&mut self, source: &Self) {\n self.input_tokens += source.input_tokens;\n self.output_tokens += source.output_tokens;\ndiff --git a/lib/crates/fabro-workflow/src/billing_rollup.rs b/lib/crates/fabro-workflow/src/billing_rollup.rs\nindex a0dac337c..a796258a3 100644\n--- a/lib/crates/fabro-workflow/src/billing_rollup.rs\n+++ b/lib/crates/fabro-workflow/src/billing_rollup.rs\n@@ -1,14 +1,29 @@\n+use std::borrow::Cow;\n use std::collections::HashMap;\n \n use fabro_model::Catalog;\n use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, StageProjection};\n \n-fn stage_priced_usage(catalog: Option<&Catalog>, stage: &StageProjection) -> BilledTokenCounts {\n- let mut usage = stage.usage.clone();\n- if let (Some(catalog), Some(model)) = (catalog, stage.model.as_ref()) {\n- usage.ensure_priced(catalog, model);\n+fn stage_usage_with_cost<'a>(\n+ catalog: Option<&Catalog>,\n+ stage: &'a StageProjection,\n+) -> Cow<'a, BilledTokenCounts> {\n+ let Some(catalog) = catalog else {\n+ return Cow::Borrowed(&stage.usage);\n+ };\n+ let Some(model) = stage.model.as_ref() else {\n+ return Cow::Borrowed(&stage.usage);\n+ };\n+ if stage.usage.total_usd_micros.is_some() {\n+ return Cow::Borrowed(&stage.usage);\n }\n- usage\n+\n+ let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {\n+ return Cow::Borrowed(&stage.usage);\n+ };\n+ let mut usage = stage.usage.clone();\n+ usage.total_usd_micros = Some(total_usd_micros);\n+ Cow::Owned(usage)\n }\n \n #[derive(Debug, Clone, PartialEq)]\n@@ -58,8 +73,9 @@ pub fn billing_rollup_from_projection(\n if is_boundary_stage(projection, stage_id.node_id()) {\n continue;\n }\n- let priced = stage_priced_usage(catalog, stage);\n- if stage.completion.is_none() && stage.duration_ms.is_none() && priced.is_zero() {\n+ let usage = stage_usage_with_cost(catalog, stage);\n+ let usage = usage.as_ref();\n+ if stage.completion.is_none() && stage.duration_ms.is_none() && usage.is_zero() {\n continue;\n }\n \n@@ -81,10 +97,10 @@ pub fn billing_rollup_from_projection(\n runtime_ms = runtime_ms.saturating_add(duration_ms);\n }\n \n- if !priced.is_zero() {\n+ if !usage.is_zero() {\n billed_visit_count += 1;\n- row.billing.add_counts(&priced);\n- totals.add_counts(&priced);\n+ row.billing.add_counts(usage);\n+ totals.add_counts(usage);\n \n if let Some(model) = &stage.model {\n row.model = Some(model.clone());\n@@ -97,7 +113,7 @@ pub fn billing_rollup_from_projection(\n billing: BilledTokenCounts::default(),\n });\n model_entry.stages += 1;\n- model_entry.billing.add_counts(&priced);\n+ model_entry.billing.add_counts(usage);\n }\n }\n }\n",
|
||||
"summary": {
|
||||
"files_changed": 5,
|
||||
"additions": 118,
|
||||
"deletions": 17
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"seq": 0,
|
||||
"checkpoint": {
|
||||
"timestamp": "2026-05-21T23:20:05.982382Z",
|
||||
"current_node": "verify",
|
||||
"completed_nodes": [
|
||||
"start",
|
||||
"toolchain",
|
||||
"preflight_compile",
|
||||
"preflight_lint",
|
||||
"implement",
|
||||
"simplify_opus",
|
||||
"simplify_gpt",
|
||||
"verify"
|
||||
],
|
||||
"node_retries": {},
|
||||
"context_values": {
|
||||
"graph.rankdir": "LR",
|
||||
"internal.node_visit_count": 1,
|
||||
"thread.toolchain.current_node": "preflight_compile",
|
||||
"graph.goal": "# Plan: Compute LLM cost on-read for in-flight stages\n\n## Context\n\nOn the run billing page (`/runs/{id}/billing`), an active stage shows token\nusage but no dollar cost — cost renders as `—` until the stage completes.\n\nRoot cause: while a stage runs, `AgentMessage` events carry usage built by\n`billed_token_counts_from_llm` (`fabro-workflow/src/outcome.rs:43`), which\nhard-codes `total_usd_micros: None`. Dollar cost is only computed by\n`billed_model_usage_from_llm` (`outcome.rs:14`) — which needs the pricing\n`Catalog` — and that runs only on `StageCompleted`/`PromptCompleted`/`StageFailed`.\nSo an in-flight stage's `StageProjection.usage.total_usd_micros` stays `None`.\n\nFix: price stages whose cost is `None` when the billing rollup is built for a\nread request, using the model + token counts already in the projection. The\nwire contract is unchanged (`total_usd_micros` is already nullable everywhere)\nand the frontend already renders whatever value comes back — no UI change.\n\n## Decisions\n\n- **Price any stage with `total_usd_micros == None`**, not just in-flight ones.\n Completed stages with unpriceable providers (`BillingPolicy::None`) return\n `None` again — harmless; no need to thread `StageState`.\n- **No \"estimated\" label.** Cost-so-far is exact for tokens consumed so far,\n matching the already-unlabeled live token counts and ticking runtime.\n- **Aggregate billing stays finalized-only.** The `BillingAccumulator` call\n sites pass `None` so a run's running estimate is never folded into org-wide\n totals (avoids double-count when the run later finalizes).\n- Per-stage rows, `totals`, and `by_model` are all priced from the same source\n so the billing page stays internally consistent.\n\n## Changes\n\n### 1. `lib/crates/fabro-model/src/billing.rs`\n\n- Add `BilledTokenCounts::token_counts(&self) -> TokenCounts` — drops\n `total_tokens`/`total_usd_micros`, keeps the five disjoint buckets.\n- Add `Catalog::price_tokens(&self, model: &ModelRef, tokens: &TokenCounts) -> Option<i64>`\n next to `pricing_for`/`billing_facts_for`. Body mirrors the cost lines of\n `billed_model_usage_from_llm`: build `ModelBillingFacts` via\n `billing_facts_for`, assemble `ModelBillingInput { ModelUsage { model, tokens }, facts }`,\n then `pricing_for(model).and_then(|p| p.bill(&input)).map(|a| a.0)`. Returns\n `None` when the provider has no billing policy.\n\n### 2. `lib/crates/fabro-workflow/src/billing_rollup.rs`\n\n- Change signature to\n `billing_rollup_from_projection(projection: &RunProjection, catalog: Option<&Catalog>)`.\n- Add a module-private helper `stage_usage_with_cost(catalog, stage) -> BilledTokenCounts`:\n clone `stage.usage`; if `total_usd_micros.is_none()` and both `catalog` and\n `stage.model` are present, set it via `catalog.price_tokens(model, &usage.token_counts())`.\n- In the loop, compute `priced` once per stage and use it in place of\n `&stage.usage` for the `is_zero` check, `row.billing.add_counts`,\n `totals.add_counts`, and `model_entry.billing.add_counts`.\n- Update the existing tests to pass `None`; add one new test: an in-flight\n stage (no `completion`, non-zero `usage` with `total_usd_micros: None`, a\n builtin `model`) yields `Some(..)` cost on the stage row and in `totals` when\n called with `Some(Catalog::builtin())`.\n\n### 3. Call sites of `billing_rollup_from_projection`\n\n- `lib/crates/fabro-server/src/server/handler/billing.rs:82` — bind\n `let catalog = state.catalog();` (returns `Arc<Catalog>`) and pass\n `Some(&catalog)`.\n- `lib/crates/fabro-server/src/server.rs` (2 aggregate-billing sites) — pass `None`.\n- `lib/crates/fabro-workflow/src/pipeline/finalize.rs` (4 sites) — pass `None`\n (stages already priced at completion; pricing would be a no-op anyway).\n\nNo changes to `fabro-api.yaml`, the generated clients, or `apps/fabro-web`.\n\n## Out of scope / known limitation\n\nIn-flight **prompt** stages have no `model` until `PromptCompleted` (only\n`AgentMessage` sets `stage.model` mid-run), so they still show `—` while\nrunning. Acceptable: prompt stages are a single short LLM call. The bug report\nconcerns agent stages, where `model` is available.\n\n## Verification\n\n- `cargo nextest run -p fabro-workflow billing_rollup` — new + updated unit tests pass.\n- `cargo nextest run -p fabro-server billing` — handler conformance still passes.\n- `cargo build --workspace` and `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings`.\n- Manual: `fabro server start` + `cd apps/fabro-web && bun run dev`, start a\n workflow with an agent stage, open `/runs/{id}/billing` mid-run — the active\n stage row and totals show a non-`—` dollar amount that grows with tokens.\n",
|
||||
"internal.retry_count.verify": 0,
|
||||
"thread.start.current_node": "toolchain",
|
||||
"response.simplify_gpt": "Implemented review cleanup on top of the billing cost-on-read changes.\n\nWhat changed:\n- Removed `BilledTokenCounts::ensure_priced`, which was an extra thin helper only used by rollup code.\n- Renamed `to_token_counts()` to `token_counts()` to match the requested API and nearby accessor style.\n- Updated `billing_rollup` pricing helper to avoid cloning stage usage unless a missing cost can actually be priced:\n - borrows existing usage for finalized/already-priced/no-catalog/no-model/unpriceable cases\n - only clones when setting a computed `total_usd_micros`\n\nReview outcome:\n- The parallel review agents did not return actionable findings.\n- I applied the above cleanup for reduced abstraction and less per-read work.\n\nVerification passed:\n- `cargo nextest run -p fabro-workflow billing_rollup`\n- `cargo nextest run -p fabro-server billing`\n- `cargo +nightly-2026-04-14 clippy -q -p fabro-model -p fabro-workflow -p fabro-server --all-targets -- -D warnings`\n\nCurrent working tree has modifications in:\n- `lib/crates/fabro-model/src/billing.rs`\n- `lib/crates/fabro-workflow/src/billing_rollup.rs`",
|
||||
"thread.implement.current_node": "simplify_opus",
|
||||
|
|
@ -1088,7 +1294,7 @@
|
|||
"failure_signature": "",
|
||||
"internal.retry_count.simplify_gpt": 0,
|
||||
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
||||
"current_node": "simplify_gpt",
|
||||
"current_node": "verify",
|
||||
"thread.simplify_opus.current_node": "simplify_gpt",
|
||||
"internal.retry_count.preflight_lint": 0,
|
||||
"thread.preflight_lint.current_node": "implement",
|
||||
|
|
@ -1097,12 +1303,13 @@
|
|||
"thread.preflight_compile.current_node": "preflight_lint",
|
||||
"response.simplify_opus": "All clean. Here's the summary:\n\n## Summary\n\nThree review agents (reuse, quality, efficiency) ran in parallel against the diff. Aggregated findings:\n\n### Fixed\n\n1. **Renamed `BilledTokenCounts::token_counts()` → `to_token_counts()`** (`fabro-model/src/billing.rs`)\n - Quality finding: original name read like a field accessor and hid the fact that the conversion is lossy (drops `total_tokens` and `total_usd_micros`). The `to_` prefix is the established Rust convention for lossy conversions.\n\n2. **Moved pricing logic into `BilledTokenCounts::ensure_priced(catalog, model)`** (`fabro-model/src/billing.rs` + `fabro-workflow/src/billing_rollup.rs`)\n - Quality finding: the previous `stage_usage_with_cost` helper cloned `BilledTokenCounts` across a crate boundary and directly poked the private `total_usd_micros` field, leaking knowledge of how to price a `BilledTokenCounts` out of `fabro-model`. The new method encapsulates that logic next to `add_billed_usage` / `replace_with_billed_usage`, and the rollup-side helper is now a 3-line wrapper.\n\n### Considered and skipped\n\n- **Reuse #1 (refactor `billed_model_usage_from_llm` to use `price_tokens`)** — the \"duplicated\" pricing tail is 3 lines and the surrounding error-vs-option shapes don't compose cleanly. Refactoring adds more indirection than it removes.\n- **`Option<&Catalog>` parameter sprawl** — 6 of 7 call sites pass `None`, but splitting into two functions doubles the public surface for marginal benefit; the single signature is fine.\n- **`pricing_for` memoization across stages** — real but minor (a few `String` clones per stage on a polled endpoint). Worth doing if the path ever shows up in profiles; not justified at the cost of HashMap setup for this fix.\n- **`Cow::Borrowed` for the usage clone** — `BilledTokenCounts` is ~64 bytes of POD; no heap allocation.\n- **Bare `i64` return from `price_tokens`** — matches the established `total_usd_micros: Option<i64>` convention used throughout the codebase.\n\n### Verified\n\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4/4 pass (including the new in-flight pricing test).\n- `cargo nextest run -p fabro-model billed_token_counts` — 5/5 pass.\n- `cargo nextest run -p fabro-server billing` — 7/7 pass.\n- `cargo build --workspace` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.",
|
||||
"outcome": "succeeded",
|
||||
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
||||
"internal.thread_id": "simplify_opus",
|
||||
"command.output": "blob://sha256/a43b1e2af7e400c4bfa32f53bd50060c91a0c8416e3cb8e8be1341fa246d22b4",
|
||||
"internal.thread_id": "simplify_gpt",
|
||||
"internal.retry_count.toolchain": 0,
|
||||
"internal.work_dir": "/home/daytona/workspace/fabro",
|
||||
"internal.retry_count.simplify_opus": 0,
|
||||
"failure_class": "",
|
||||
"thread.simplify_gpt.current_node": "verify",
|
||||
"response.implement": "All changes implemented as specified in the plan:\n\n**1. `lib/crates/fabro-model/src/billing.rs`** ✅\n- Added `BilledTokenCounts::token_counts(&self) -> TokenCounts` (drops `total_tokens`/`total_usd_micros`).\n- Added `Catalog::price_tokens(&self, model, tokens) -> Option<i64>` that mirrors the cost lines of `billed_model_usage_from_llm`.\n\n**2. `lib/crates/fabro-workflow/src/billing_rollup.rs`** ✅\n- Changed `billing_rollup_from_projection` signature to take `catalog: Option<&Catalog>`.\n- Added private helper `stage_usage_with_cost` that clones `stage.usage` and prices it when `total_usd_micros.is_none()` and both `catalog` and `stage.model` are available.\n- Uses `priced` in place of `&stage.usage` for the `is_zero` check, `row.billing.add_counts`, `totals.add_counts`, and `model_entry.billing.add_counts`.\n- Updated existing tests to pass `None`; added a new `rollup_prices_in_flight_stage_usage_using_catalog` test (in-flight stage with no completion, non-zero usage, builtin model → `Some(..)` cost on stage row, totals, and `by_model`).\n\n**3. Call sites** ✅\n- `fabro-server/src/server/handler/billing.rs:82` — binds `let catalog = state.catalog();` and passes `Some(&catalog)`.\n- `fabro-server/src/server.rs` (2 aggregate-billing sites) — pass `None`.\n- `fabro-workflow/src/pipeline/finalize.rs` (4 sites) — pass `None`.\n\n**Verification** ✅\n- `cargo nextest run -p fabro-workflow billing_rollup` — 4 tests pass (new + 3 updated).\n- `cargo nextest run -p fabro-server billing` — 7 tests pass.\n- `cargo nextest run -p fabro-model billing` — 23 tests pass.\n- `cargo check --workspace --all-targets` — clean.\n- `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` — clean.\n- `cargo +nightly-2026-04-14 fmt --check --all` — clean.\n- Broader test sweep on `fabro-workflow` + `fabro-server`: 1674 passing; the 2 unrelated `*graph*svg*` failures are pre-existing graphviz subprocess env issues (confirmed by stashing my changes and reproducing).",
|
||||
"internal.run_id": "01KS6AW28FZVV4M2EHBJA7JMNP",
|
||||
"internal.retry_count.preflight_compile": 0,
|
||||
|
|
@ -1164,6 +1371,14 @@
|
|||
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
||||
"usage": null
|
||||
},
|
||||
"verify": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
"command.output": "blob://sha256/a43b1e2af7e400c4bfa32f53bd50060c91a0c8416e3cb8e8be1341fa246d22b4"
|
||||
},
|
||||
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
||||
"usage": null
|
||||
},
|
||||
"toolchain": {
|
||||
"status": "succeeded",
|
||||
"context_updates": {
|
||||
|
|
@ -1243,14 +1458,15 @@
|
|||
}
|
||||
}
|
||||
},
|
||||
"next_node_id": "verify",
|
||||
"next_node_id": "fmt",
|
||||
"node_visits": {
|
||||
"implement": 1,
|
||||
"simplify_opus": 1,
|
||||
"simplify_gpt": 1,
|
||||
"preflight_compile": 1,
|
||||
"preflight_lint": 1,
|
||||
"start": 1,
|
||||
"implement": 1,
|
||||
"simplify_gpt": 1,
|
||||
"verify": 1,
|
||||
"preflight_lint": 1,
|
||||
"toolchain": 1
|
||||
}
|
||||
},
|
||||
|
|
@ -1278,44 +1494,6 @@
|
|||
"superseded_by": null,
|
||||
"pending_interviews": {},
|
||||
"stages": {
|
||||
"implement@1": {
|
||||
"first_event_seq": 50,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": "Stage completed: implement",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:59:35.631657Z"
|
||||
},
|
||||
"provider_used": {
|
||||
"mode": "agent",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-7"
|
||||
},
|
||||
"diff": null,
|
||||
"script_invocation": null,
|
||||
"script_timing": null,
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"started_at": "2026-05-21T22:46:11.047830Z",
|
||||
"handler": "agent",
|
||||
"duration_ms": 804582,
|
||||
"usage": {
|
||||
"input_tokens": 92896,
|
||||
"output_tokens": 16017,
|
||||
"total_tokens": 4900766,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 4690781,
|
||||
"cache_write_tokens": 101072,
|
||||
"total_usd_micros": 3841995
|
||||
},
|
||||
"model": {
|
||||
"provider": "anthropic",
|
||||
"model_id": "claude-opus-4-7"
|
||||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"preflight_compile@1": {
|
||||
"first_event_seq": 30,
|
||||
"prompt": null,
|
||||
|
|
@ -1359,78 +1537,6 @@
|
|||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"toolchain@1": {
|
||||
"first_event_seq": 20,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:36:09.940422Z"
|
||||
},
|
||||
"provider_used": null,
|
||||
"diff": null,
|
||||
"script_invocation": {
|
||||
"script": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"command": "exec 2>&1\ncommand -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"language": "shell"
|
||||
},
|
||||
"script_timing": {
|
||||
"output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c",
|
||||
"exit_code": 0,
|
||||
"duration_ms": 1663,
|
||||
"termination": "exited",
|
||||
"output_bytes": 36,
|
||||
"live_streaming": true
|
||||
},
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"output_bytes": 36,
|
||||
"live_streaming": true,
|
||||
"termination": "exited",
|
||||
"started_at": "2026-05-21T22:36:08.268854Z",
|
||||
"handler": "command",
|
||||
"duration_ms": 1671,
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"cache_write_tokens": 0
|
||||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"start@1": {
|
||||
"first_event_seq": 16,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": null,
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:36:08.268697Z"
|
||||
},
|
||||
"provider_used": null,
|
||||
"diff": null,
|
||||
"script_invocation": null,
|
||||
"script_timing": null,
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"started_at": "2026-05-21T22:36:08.268606Z",
|
||||
"handler": "start",
|
||||
"duration_ms": 0,
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"cache_write_tokens": 0
|
||||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"preflight_lint@1": {
|
||||
"first_event_seq": 40,
|
||||
"prompt": null,
|
||||
|
|
@ -1474,6 +1580,62 @@
|
|||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"start@1": {
|
||||
"first_event_seq": 16,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": null,
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:36:08.268697Z"
|
||||
},
|
||||
"provider_used": null,
|
||||
"diff": null,
|
||||
"script_invocation": null,
|
||||
"script_timing": null,
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"started_at": "2026-05-21T22:36:08.268606Z",
|
||||
"handler": "start",
|
||||
"duration_ms": 0,
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"cache_write_tokens": 0
|
||||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"verify@1": {
|
||||
"first_event_seq": 808,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": null,
|
||||
"provider_used": null,
|
||||
"diff": null,
|
||||
"script_invocation": {
|
||||
"script": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
||||
"command": "exec 2>&1\ncargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
||||
"language": "shell"
|
||||
},
|
||||
"script_timing": null,
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"started_at": "2026-05-21T23:16:25.739901Z",
|
||||
"handler": "command",
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"cache_write_tokens": 0
|
||||
},
|
||||
"state": "running"
|
||||
},
|
||||
"simplify_opus@1": {
|
||||
"first_event_seq": 255,
|
||||
"prompt": null,
|
||||
|
|
@ -1516,7 +1678,12 @@
|
|||
"first_event_seq": 484,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": "Stage completed: simplify_gpt",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T23:16:18.947478Z"
|
||||
},
|
||||
"provider_used": {
|
||||
"mode": "agent",
|
||||
"provider": "openai",
|
||||
|
|
@ -1529,6 +1696,7 @@
|
|||
"output": null,
|
||||
"started_at": "2026-05-21T23:10:20.517687Z",
|
||||
"handler": "agent",
|
||||
"duration_ms": 358429,
|
||||
"usage": {
|
||||
"input_tokens": 89620,
|
||||
"output_tokens": 7291,
|
||||
|
|
@ -1542,7 +1710,88 @@
|
|||
"provider": "openai",
|
||||
"model_id": "gpt-5.5"
|
||||
},
|
||||
"state": "running"
|
||||
"state": "succeeded"
|
||||
},
|
||||
"implement@1": {
|
||||
"first_event_seq": 50,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": "Stage completed: implement",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:59:35.631657Z"
|
||||
},
|
||||
"provider_used": {
|
||||
"mode": "agent",
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-7"
|
||||
},
|
||||
"diff": null,
|
||||
"script_invocation": null,
|
||||
"script_timing": null,
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"started_at": "2026-05-21T22:46:11.047830Z",
|
||||
"handler": "agent",
|
||||
"duration_ms": 804582,
|
||||
"usage": {
|
||||
"input_tokens": 92896,
|
||||
"output_tokens": 16017,
|
||||
"total_tokens": 4900766,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 4690781,
|
||||
"cache_write_tokens": 101072,
|
||||
"total_usd_micros": 3841995
|
||||
},
|
||||
"model": {
|
||||
"provider": "anthropic",
|
||||
"model_id": "claude-opus-4-7"
|
||||
},
|
||||
"state": "succeeded"
|
||||
},
|
||||
"toolchain@1": {
|
||||
"first_event_seq": 20,
|
||||
"prompt": null,
|
||||
"response": null,
|
||||
"completion": {
|
||||
"outcome": "succeeded",
|
||||
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T22:36:09.940422Z"
|
||||
},
|
||||
"provider_used": null,
|
||||
"diff": null,
|
||||
"script_invocation": {
|
||||
"script": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"command": "exec 2>&1\ncommand -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
||||
"language": "shell"
|
||||
},
|
||||
"script_timing": {
|
||||
"output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c",
|
||||
"exit_code": 0,
|
||||
"duration_ms": 1663,
|
||||
"termination": "exited",
|
||||
"output_bytes": 36,
|
||||
"live_streaming": true
|
||||
},
|
||||
"parallel_results": null,
|
||||
"output": null,
|
||||
"output_bytes": 36,
|
||||
"live_streaming": true,
|
||||
"termination": "exited",
|
||||
"started_at": "2026-05-21T22:36:08.268854Z",
|
||||
"handler": "command",
|
||||
"duration_ms": 1671,
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
"reasoning_tokens": 0,
|
||||
"cache_read_tokens": 0,
|
||||
"cache_write_tokens": 0
|
||||
},
|
||||
"state": "succeeded"
|
||||
}
|
||||
}
|
||||
}
|
||||
105
stages/007-simplify_gpt@1/diff.patch
Normal file
105
stages/007-simplify_gpt@1/diff.patch
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
diff --git a/lib/crates/fabro-model/src/billing.rs b/lib/crates/fabro-model/src/billing.rs
|
||||
index a3be9f470..f06cad06a 100644
|
||||
--- a/lib/crates/fabro-model/src/billing.rs
|
||||
+++ b/lib/crates/fabro-model/src/billing.rs
|
||||
@@ -361,7 +361,7 @@ impl BilledTokenCounts {
|
||||
/// Returns the five disjoint per-call token buckets, dropping the derived
|
||||
/// `total_tokens` sum and the optional `total_usd_micros` cost.
|
||||
#[must_use]
|
||||
- pub fn to_token_counts(&self) -> TokenCounts {
|
||||
+ pub fn token_counts(&self) -> TokenCounts {
|
||||
TokenCounts {
|
||||
input_tokens: self.input_tokens,
|
||||
output_tokens: self.output_tokens,
|
||||
@@ -371,17 +371,6 @@ impl BilledTokenCounts {
|
||||
}
|
||||
}
|
||||
|
||||
- /// Fills in `total_usd_micros` from catalog pricing when it is missing.
|
||||
- ///
|
||||
- /// No-op when the cost is already known. Used by read-side rollups so
|
||||
- /// in-flight stages can show an exact cost for the tokens consumed so far
|
||||
- /// without mutating the underlying projection.
|
||||
- pub fn ensure_priced(&mut self, catalog: &Catalog, model: &ModelRef) {
|
||||
- if self.total_usd_micros.is_none() {
|
||||
- self.total_usd_micros = catalog.price_tokens(model, &self.to_token_counts());
|
||||
- }
|
||||
- }
|
||||
-
|
||||
pub fn add_counts(&mut self, source: &Self) {
|
||||
self.input_tokens += source.input_tokens;
|
||||
self.output_tokens += source.output_tokens;
|
||||
diff --git a/lib/crates/fabro-workflow/src/billing_rollup.rs b/lib/crates/fabro-workflow/src/billing_rollup.rs
|
||||
index a0dac337c..a796258a3 100644
|
||||
--- a/lib/crates/fabro-workflow/src/billing_rollup.rs
|
||||
+++ b/lib/crates/fabro-workflow/src/billing_rollup.rs
|
||||
@@ -1,14 +1,29 @@
|
||||
+use std::borrow::Cow;
|
||||
use std::collections::HashMap;
|
||||
|
||||
use fabro_model::Catalog;
|
||||
use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, StageProjection};
|
||||
|
||||
-fn stage_priced_usage(catalog: Option<&Catalog>, stage: &StageProjection) -> BilledTokenCounts {
|
||||
- let mut usage = stage.usage.clone();
|
||||
- if let (Some(catalog), Some(model)) = (catalog, stage.model.as_ref()) {
|
||||
- usage.ensure_priced(catalog, model);
|
||||
+fn stage_usage_with_cost<'a>(
|
||||
+ catalog: Option<&Catalog>,
|
||||
+ stage: &'a StageProjection,
|
||||
+) -> Cow<'a, BilledTokenCounts> {
|
||||
+ let Some(catalog) = catalog else {
|
||||
+ return Cow::Borrowed(&stage.usage);
|
||||
+ };
|
||||
+ let Some(model) = stage.model.as_ref() else {
|
||||
+ return Cow::Borrowed(&stage.usage);
|
||||
+ };
|
||||
+ if stage.usage.total_usd_micros.is_some() {
|
||||
+ return Cow::Borrowed(&stage.usage);
|
||||
}
|
||||
- usage
|
||||
+
|
||||
+ let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
|
||||
+ return Cow::Borrowed(&stage.usage);
|
||||
+ };
|
||||
+ let mut usage = stage.usage.clone();
|
||||
+ usage.total_usd_micros = Some(total_usd_micros);
|
||||
+ Cow::Owned(usage)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
@@ -58,8 +73,9 @@ pub fn billing_rollup_from_projection(
|
||||
if is_boundary_stage(projection, stage_id.node_id()) {
|
||||
continue;
|
||||
}
|
||||
- let priced = stage_priced_usage(catalog, stage);
|
||||
- if stage.completion.is_none() && stage.duration_ms.is_none() && priced.is_zero() {
|
||||
+ let usage = stage_usage_with_cost(catalog, stage);
|
||||
+ let usage = usage.as_ref();
|
||||
+ if stage.completion.is_none() && stage.duration_ms.is_none() && usage.is_zero() {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -81,10 +97,10 @@ pub fn billing_rollup_from_projection(
|
||||
runtime_ms = runtime_ms.saturating_add(duration_ms);
|
||||
}
|
||||
|
||||
- if !priced.is_zero() {
|
||||
+ if !usage.is_zero() {
|
||||
billed_visit_count += 1;
|
||||
- row.billing.add_counts(&priced);
|
||||
- totals.add_counts(&priced);
|
||||
+ row.billing.add_counts(usage);
|
||||
+ totals.add_counts(usage);
|
||||
|
||||
if let Some(model) = &stage.model {
|
||||
row.model = Some(model.clone());
|
||||
@@ -97,7 +113,7 @@ pub fn billing_rollup_from_projection(
|
||||
billing: BilledTokenCounts::default(),
|
||||
});
|
||||
model_entry.stages += 1;
|
||||
- model_entry.billing.add_counts(&priced);
|
||||
+ model_entry.billing.add_counts(usage);
|
||||
}
|
||||
}
|
||||
}
|
||||
6
stages/007-simplify_gpt@1/status.json
Normal file
6
stages/007-simplify_gpt@1/status.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"outcome": "succeeded",
|
||||
"notes": "Stage completed: simplify_gpt",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-21T23:16:18.947478Z"
|
||||
}
|
||||
5
stages/008-verify@1/script_invocation.json
Normal file
5
stages/008-verify@1/script_invocation.json
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
{
|
||||
"script": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
||||
"command": "exec 2>&1\ncargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
||||
"language": "shell"
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue