Merge remote-tracking branch 'origin/main' into fix/small-default-skips-providers-without-small-model

# Conflicts:
#	lib/apps/fabro-server/src/test_support.rs
This commit is contained in:
Bryan Helmkamp 2026-07-28 17:06:20 -04:00
commit df0bd58819
No known key found for this signature in database
149 changed files with 7556 additions and 1634 deletions

View file

@ -11,10 +11,6 @@ leak-timeout = "500ms"
filter = "package(fabro-server)"
slow-timeout = { period = "5s", terminate-after = 4 }
[[profile.default.overrides]]
filter = "package(fabro-server) & test(all_spec_routes_are_routable)"
slow-timeout = { period = "15s", terminate-after = 4 }
[[profile.default.overrides]]
filter = "package(fabro-workflow)"
slow-timeout = { period = "2s", terminate-after = 3 }

103
Cargo.lock generated
View file

@ -2239,7 +2239,7 @@ dependencies = [
[[package]]
name = "fabro-acp"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"agent-client-protocol",
"agent-client-protocol-tokio",
@ -2258,7 +2258,7 @@ dependencies = [
[[package]]
name = "fabro-agent"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -2304,7 +2304,7 @@ dependencies = [
[[package]]
name = "fabro-api"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"chrono",
"fabro-automation",
@ -2327,7 +2327,7 @@ dependencies = [
[[package]]
name = "fabro-auth"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -2352,7 +2352,7 @@ dependencies = [
[[package]]
name = "fabro-automation"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -2371,11 +2371,11 @@ dependencies = [
[[package]]
name = "fabro-build-support"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
[[package]]
name = "fabro-checkpoint"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"chrono",
"fabro-config",
@ -2391,7 +2391,7 @@ dependencies = [
[[package]]
name = "fabro-cli"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"assert_cmd",
@ -2493,7 +2493,7 @@ dependencies = [
[[package]]
name = "fabro-client"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"bytes",
@ -2522,7 +2522,7 @@ dependencies = [
[[package]]
name = "fabro-config"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -2552,7 +2552,7 @@ dependencies = [
[[package]]
name = "fabro-core"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"async-trait",
"fabro-types",
@ -2567,7 +2567,7 @@ dependencies = [
[[package]]
name = "fabro-db"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -2579,7 +2579,7 @@ dependencies = [
[[package]]
name = "fabro-dev"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"assert_cmd",
@ -2598,7 +2598,7 @@ dependencies = [
[[package]]
name = "fabro-dump"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"bytes",
@ -2612,7 +2612,7 @@ dependencies = [
[[package]]
name = "fabro-environment"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -2634,7 +2634,7 @@ dependencies = [
[[package]]
name = "fabro-github"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"base64",
@ -2656,7 +2656,7 @@ dependencies = [
[[package]]
name = "fabro-graphviz"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"fabro-types",
@ -2671,7 +2671,7 @@ dependencies = [
[[package]]
name = "fabro-hooks"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"async-trait",
"fabro-agent",
@ -2694,7 +2694,7 @@ dependencies = [
[[package]]
name = "fabro-http"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"fabro-static",
"http 1.4.0",
@ -2704,7 +2704,7 @@ dependencies = [
[[package]]
name = "fabro-install"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"base64",
@ -2723,7 +2723,7 @@ dependencies = [
[[package]]
name = "fabro-interview"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"async-trait",
"dialoguer",
@ -2738,7 +2738,7 @@ dependencies = [
[[package]]
name = "fabro-llm"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -2779,7 +2779,7 @@ dependencies = [
[[package]]
name = "fabro-macros"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"clap",
"fabro-options-metadata",
@ -2790,7 +2790,7 @@ dependencies = [
[[package]]
name = "fabro-manifest"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"fabro-api",
@ -2808,7 +2808,7 @@ dependencies = [
[[package]]
name = "fabro-mcp"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"axum",
@ -2828,7 +2828,7 @@ dependencies = [
[[package]]
name = "fabro-mcp-server"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -2855,7 +2855,7 @@ dependencies = [
[[package]]
name = "fabro-mcp-store"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"chrono",
"fabro-db",
@ -2873,7 +2873,7 @@ dependencies = [
[[package]]
name = "fabro-model"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"fabro-static",
"http 1.4.0",
@ -2889,7 +2889,7 @@ dependencies = [
[[package]]
name = "fabro-oauth"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"axum",
@ -2911,7 +2911,7 @@ dependencies = [
[[package]]
name = "fabro-options-metadata"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"serde",
"serde_json",
@ -2919,7 +2919,7 @@ dependencies = [
[[package]]
name = "fabro-proc"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"cc",
"libc",
@ -2928,7 +2928,7 @@ dependencies = [
[[package]]
name = "fabro-redact"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"aho-corasick",
"ref-cast",
@ -2944,7 +2944,7 @@ dependencies = [
[[package]]
name = "fabro-sandbox"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -2988,7 +2988,7 @@ dependencies = [
[[package]]
name = "fabro-server"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -3081,7 +3081,7 @@ dependencies = [
[[package]]
name = "fabro-slack"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"fabro-http",
"fabro-interview",
@ -3103,18 +3103,18 @@ dependencies = [
[[package]]
name = "fabro-spa"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"rust-embed",
]
[[package]]
name = "fabro-static"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
[[package]]
name = "fabro-store"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"async-trait",
"bytes",
@ -3144,7 +3144,7 @@ dependencies = [
[[package]]
name = "fabro-telemetry"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"base64",
@ -3170,7 +3170,7 @@ dependencies = [
[[package]]
name = "fabro-template"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"fabro-types",
@ -3184,7 +3184,7 @@ dependencies = [
[[package]]
name = "fabro-test"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"assert_cmd",
@ -3209,7 +3209,7 @@ dependencies = [
[[package]]
name = "fabro-tool"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -3230,7 +3230,7 @@ dependencies = [
[[package]]
name = "fabro-tracker"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-trait",
@ -3244,7 +3244,7 @@ dependencies = [
[[package]]
name = "fabro-types"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"chrono",
"clap",
@ -3259,6 +3259,7 @@ dependencies = [
"shlex",
"strum 0.28.0",
"tempfile",
"thiserror 2.0.18",
"toml 0.8.23",
"ulid",
"url",
@ -3266,7 +3267,7 @@ dependencies = [
[[package]]
name = "fabro-util"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"console 0.15.11",
@ -3289,7 +3290,7 @@ dependencies = [
[[package]]
name = "fabro-validate"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"fabro-acp",
"fabro-graphviz",
@ -3302,7 +3303,7 @@ dependencies = [
[[package]]
name = "fabro-variable"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -3319,7 +3320,7 @@ dependencies = [
[[package]]
name = "fabro-vault"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"chrono",
@ -3338,7 +3339,7 @@ dependencies = [
[[package]]
name = "fabro-workflow"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"assert_cmd",
@ -8502,7 +8503,7 @@ dependencies = [
[[package]]
name = "twin-github"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"axum",
"base64",
@ -8521,7 +8522,7 @@ dependencies = [
[[package]]
name = "twin-openai"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
dependencies = [
"anyhow",
"async-stream",

View file

@ -11,7 +11,7 @@ resolver = "2"
[workspace.package]
edition = "2021"
version = "0.305.0-nightly.3"
version = "0.308.0-nightly.0"
license = "MIT"
[workspace.dependencies]

View file

@ -4,9 +4,14 @@ import { SWRConfig } from "swr";
import {
type ApiQuestion,
QuestionType,
ReviewTargetKind,
} from "@qltysh/fabro-api-client";
import { InterviewDock } from "./interview-dock";
import {
contextPreview,
InterviewDock,
shouldStackOptions,
} from "./interview-dock";
import { displayLabel } from "./interview-label";
import { generatedAxios } from "../lib/api-client";
@ -74,6 +79,58 @@ describe("InterviewDock", () => {
expect(text).toContain("Awaiting input");
});
test("renders the review target as the link in the question", () => {
const url =
"https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef";
const tree = render(
<InterviewDock
runId="run-1"
questions={[
makeQuestion({
text: "Review the Quarry review exercise document, then choose the next action.",
review_target: {
label: "Quarry review exercise",
url,
kind: ReviewTargetKind.DOCUMENT,
},
}),
]}
/>,
);
expect(textContent(tree.toJSON())).toContain(
"Review the Quarry review exercise document, then choose the next action.",
);
const links = tree.root.findAllByType("a");
expect(links).toHaveLength(1);
expect(links[0].props.href).toBe(url);
expect(links[0].props.target).toBe("_blank");
expect(links[0].props.rel).toBe("noopener noreferrer");
expect(links[0].props.referrerPolicy).toBe("no-referrer");
});
test("does not link an unsafe review target received from the API", () => {
const fallback = "Review the document, then choose the next action.";
const tree = render(
<InterviewDock
runId="run-1"
questions={[
makeQuestion({
text: fallback,
review_target: {
label: "Unsafe target",
url: "javascript:alert(1)",
kind: ReviewTargetKind.DOCUMENT,
},
}),
]}
/>,
);
expect(textContent(tree.toJSON())).toContain(fallback);
expect(tree.root.findAllByType("a")).toHaveLength(0);
});
test("yes/no question shows two buttons", () => {
const tree = render(
<InterviewDock runId="run-1" questions={[makeQuestion()]} />,
@ -236,6 +293,112 @@ describe("InterviewDock", () => {
expect(text).toContain("Context from preceding stage");
expect(text).toContain("1. Deploy");
});
test("context starts closed so it costs one line, not a standing panel", () => {
const question = makeQuestion({
context_display: "Plan:\n1. Deploy\n2. Verify",
});
const tree = render(
<InterviewDock runId="run-1" questions={[question]} />,
);
const details = tree.root.findAllByType("details");
expect(details).toHaveLength(1);
expect(details[0]!.props.open).toBeFalsy();
});
test("omits the question type subtitle that the answer buttons already state", () => {
const question = makeQuestion({
question_type: QuestionType.MULTIPLE_CHOICE,
options: [{ key: "A", label: "[A] Approve" }],
});
const tree = render(
<InterviewDock runId="run-1" questions={[question]} />,
);
expect(textContent(tree.toJSON())).not.toContain("Pick one");
});
test("collapsing hides the answer controls but keeps the header", () => {
const tree = render(
<InterviewDock runId="run-1" questions={[makeQuestion()]} />,
);
const toggle = tree.root.findByProps({ "aria-label": "Collapse Interview question" });
act(() => {
toggle.props.onClick();
});
const expanded = tree.root.findByProps({
"aria-label": "Expand Interview question",
});
expect(expanded.props["aria-expanded"]).toBe(false);
expect(textContent(tree.toJSON())).toContain("Awaiting input");
});
test("a queued question arrives expanded even after the panel was collapsed", () => {
const questions = [
makeQuestion({ id: "q-1", stage: "stage-a" }),
makeQuestion({ id: "q-2", stage: "stage-b" }),
];
const tree = render(<InterviewDock runId="run-1" questions={questions} />);
act(() => {
tree.root
.findByProps({ "aria-label": "Collapse Interview question" })
.props.onClick();
});
act(() => {
buttonsByText(tree)["1 more pending"]!.props.onClick();
});
const toggle = tree.root.findByProps({
"aria-label": "Collapse Interview question",
});
expect(toggle.props["aria-expanded"]).toBe(true);
});
});
describe("shouldStackOptions", () => {
test("keeps short labels as a wrapping row", () => {
expect(
shouldStackOptions([
{ key: "a", label: "Approve" },
{ key: "b", label: "Revise" },
]),
).toBe(false);
});
test("stacks once a label is too long to sit in a pill", () => {
expect(
shouldStackOptions([
{ key: "a", label: "Approve" },
{ key: "b", label: "Review complete; sync the current Markdown document" },
]),
).toBe(true);
});
test("stacks when any option carries a description", () => {
expect(
shouldStackOptions([
{ key: "a", label: "Approve", description: "Deploy the current patch" },
]),
).toBe(true);
});
});
describe("contextPreview", () => {
test("uses the first non-empty line", () => {
expect(contextPreview("\n\nPlan is ready\nSecond line")).toBe("Plan is ready");
});
test("truncates a long first line", () => {
const preview = contextPreview("x".repeat(200));
expect(preview).toHaveLength(61);
expect(preview.endsWith("…")).toBe(true);
});
test("returns an empty string for blank context", () => {
expect(contextPreview(" \n ")).toBe("");
});
});
describe("displayLabel", () => {

View file

@ -1,14 +1,8 @@
import { useCallback, useState } from "react";
import {
useCallback,
useState,
type FormEvent,
type KeyboardEvent,
} from "react";
import {
ArrowPathIcon,
ArrowRightIcon,
ArrowUturnLeftIcon,
CheckIcon,
ChevronRightIcon,
} from "@heroicons/react/20/solid";
import { QuestionType } from "@qltysh/fabro-api-client";
import type {
@ -22,16 +16,31 @@ import {
} from "../lib/mutations";
import { ApiError } from "../lib/api-client";
import { displayLabel } from "./interview-label";
import { ErrorMessage } from "./ui";
import {
ReviewTargetQuestion,
safeReviewTarget,
} from "./review-target-question";
import {
DockComposer,
RunDockShell,
DOCK_CHOICE_BUTTON,
DOCK_CHOICE_BUTTON_SELECTED,
DOCK_HEADER_BUTTON,
} from "./run-dock";
import { Spinner } from "./state";
import {
ErrorMessage,
PRIMARY_BUTTON_CLASS,
} from "./ui";
const PRIMARY_BUTTON =
"inline-flex items-center justify-center gap-1.5 rounded-lg bg-teal-500 px-3.5 py-2 text-sm font-medium text-on-primary transition-colors hover:bg-teal-300 focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500 disabled:cursor-not-allowed disabled:opacity-60 disabled:hover:bg-teal-500";
/**
* Options stack into a list once a label is long enough that a row of pills
* would wrap mid-sentence.
*/
const STACK_LABEL_LENGTH = 40;
const CHOICE_BUTTON =
"inline-flex items-center justify-center gap-1.5 rounded-lg bg-overlay px-3.5 py-2 text-sm font-medium text-fg-2 outline-1 -outline-offset-1 outline-line-strong transition-colors hover:bg-overlay-strong hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500 disabled:cursor-not-allowed disabled:opacity-60";
const CHOICE_BUTTON_SELECTED =
"inline-flex items-center justify-center gap-1.5 rounded-lg bg-teal-500/15 px-3.5 py-2 text-sm font-medium text-fg outline-1 -outline-offset-1 outline-teal-500/60 transition-colors hover:bg-teal-500/20 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500";
/** Shared by the plain question text and the review target rendering. */
const QUESTION_TEXT = "max-w-[78ch] text-base/6 font-medium text-pretty text-fg";
type SubmitInterviewAnswer = SubmitInterviewAnswerArg["answer"];
@ -51,6 +60,8 @@ export function InterviewDock({ runId, questions }: InterviewDockProps) {
const moreCount = questions.length - 1;
return (
// Keyed by question id, so a new question always arrives expanded with an
// empty composer. A collapsed panel can never silently block a run.
<InterviewQuestionDock
key={question.id}
runId={runId}
@ -76,112 +87,115 @@ function InterviewQuestionDock({
}) {
const submitMutation = useSubmitInterviewAnswer(runId);
const [error, setError] = useState<string | null>(null);
const [collapsed, setCollapsed] = useState(false);
const submitting = submitMutation.isMutating;
const reviewTarget = safeReviewTarget(question.review_target);
const submit = useCallback(
async (answer: SubmitInterviewAnswer) => {
setError(null);
try {
await submitMutation.trigger({ questionId: question.id, answer });
return true;
} catch (caught) {
setError(interviewSubmitErrorMessage(caught));
return false;
}
},
[question.id, submitMutation],
);
return (
<section aria-label="Interview question">
<DockHeader
stage={question.stage}
moreCount={moreCount}
onCycle={onCycle}
/>
<div className="space-y-5 px-5 py-4 sm:px-6">
<div>
<p className="text-pretty text-base/6 font-medium text-fg">
{question.text}
</p>
<p className="mt-1 text-xs/5 text-fg-muted">
{questionTypeLabel(question.question_type)}
</p>
</div>
{question.context_display && (
<ContextPanel text={question.context_display} />
)}
<QuestionBody
question={question}
submitting={submitting}
onSubmit={submit}
/>
{error && <ErrorMessage message={error} />}
</div>
</section>
);
}
function DockHeader({
stage,
moreCount,
onCycle,
}: {
stage: string;
moreCount: number;
onCycle: () => void;
}) {
return (
<div className="flex items-center justify-between gap-3 border-b border-line px-5 py-2.5">
<div className="flex min-w-0 items-center gap-2 text-sm">
<PulseDot />
<span className="font-medium text-fg-2">Awaiting input</span>
{stage && (
<>
<span className="text-fg-muted" aria-hidden="true">
·
</span>
<span className="truncate font-mono text-xs text-fg-3">{stage}</span>
</>
)}
</div>
{moreCount > 0 && (
<button
type="button"
onClick={onCycle}
className="inline-flex shrink-0 items-center gap-1 rounded-md bg-overlay px-2 py-1 text-xs font-medium text-fg-2 outline-1 -outline-offset-1 outline-line-strong hover:bg-overlay-strong hover:text-fg focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500"
>
<span className="tabular-nums">{moreCount}</span> more pending
<ArrowRightIcon className="size-3" aria-hidden="true" />
</button>
)}
</div>
);
}
function PulseDot() {
return (
<span className="relative flex size-2 items-center justify-center" aria-hidden="true">
<span className="absolute inline-flex size-full animate-ping rounded-full bg-amber/60" />
<span className="relative inline-flex size-2 rounded-full bg-amber" />
</span>
<RunDockShell
label="Interview question"
tone="waiting"
status="Awaiting input"
stage={question.stage}
peek={question.text}
collapsed={collapsed}
onCollapsedChange={setCollapsed}
headerActions={
moreCount > 0 && (
<button
type="button"
onClick={onCycle}
className={DOCK_HEADER_BUTTON}
>
<span className="tabular-nums">{moreCount}</span> more pending
<ArrowRightIcon className="size-3" aria-hidden="true" />
</button>
)
}
body={
<>
{reviewTarget ? (
<ReviewTargetQuestion
target={reviewTarget}
className={QUESTION_TEXT}
/>
) : (
<p className={QUESTION_TEXT}>{question.text}</p>
)}
{question.context_display && (
<ContextPanel text={question.context_display} />
)}
</>
}
actions={
<>
<QuestionBody
question={question}
submitting={submitting}
onSubmit={submit}
/>
{error && <ErrorMessage message={error} />}
</>
}
/>
);
}
/**
* Context arrives collapsed. It repeats material the operator has usually
* already read in the stage stream above, so it earns a line rather than a
* standing panel.
*/
function ContextPanel({ text }: { text: string }) {
return (
<div className="rounded-lg bg-panel-alt p-4 outline-1 -outline-offset-1 outline-line">
<p className="mb-1.5 font-mono text-[0.6875rem] tracking-wide text-fg-muted uppercase">
<details className="group rounded-lg bg-panel-alt outline-1 -outline-offset-1 outline-line">
<summary className="flex cursor-pointer list-none items-center gap-1.5 rounded-lg px-3 py-1.5 font-mono text-[0.6875rem] tracking-wide text-fg-muted uppercase transition-colors hover:text-fg-3 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500 [&::-webkit-details-marker]:hidden">
<ChevronRightIcon
className="size-3 shrink-0 transition-transform group-open:rotate-90"
aria-hidden="true"
/>
Context from preceding stage
</p>
<div className="max-h-40 overflow-y-auto text-sm/6 text-fg-2">
<span className="ml-auto truncate pl-3 font-sans text-xs tracking-normal normal-case group-open:hidden">
{contextPreview(text)}
</span>
</summary>
<div className="px-3 pb-2.5 text-sm/6 text-fg-2">
<pre className="font-sans whitespace-pre-wrap">{text}</pre>
</div>
</div>
</details>
);
}
/** First line of the context, for the collapsed summary. */
export function contextPreview(text: string): string {
let lineStart = 0;
while (lineStart < text.length) {
const newline = text.indexOf("\n", lineStart);
const lineEnd = newline === -1 ? text.length : newline;
const line = text.slice(lineStart, lineEnd).trim();
if (line) {
return line.length > 60 ? `${line.slice(0, 60).trimEnd()}…` : line;
}
if (newline === -1) break;
lineStart = newline + 1;
}
return "";
}
function QuestionBody({
question,
submitting,
@ -189,7 +203,7 @@ function QuestionBody({
}: {
question: ApiQuestion;
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
}) {
switch (question.question_type) {
case QuestionType.YES_NO:
@ -215,11 +229,10 @@ function QuestionBody({
);
case QuestionType.FREEFORM:
return (
<FreeformBody
<FreeformAnswer
submitting={submitting}
onSubmit={onSubmit}
placeholder="Write your response…"
submitLabel="Send"
/>
);
default:
@ -232,7 +245,7 @@ function YesNoBody({
onSubmit,
}: {
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
}) {
return (
<div className="flex flex-wrap items-center gap-2">
@ -242,7 +255,7 @@ function YesNoBody({
aria-label="Answer no"
disabled={submitting}
onClick={() => void onSubmit({ kind: "no" })}
className={CHOICE_BUTTON}
className={DOCK_CHOICE_BUTTON}
>
No
</button>
@ -251,9 +264,13 @@ function YesNoBody({
aria-label="Answer yes"
disabled={submitting}
onClick={() => void onSubmit({ kind: "yes" })}
className={PRIMARY_BUTTON}
className={PRIMARY_BUTTON_CLASS}
>
{submitting ? <Spinner /> : <CheckIcon className="size-4" aria-hidden="true" />}
{submitting ? (
<Spinner className="size-4" />
) : (
<CheckIcon className="size-4" aria-hidden="true" />
)}
Yes
</button>
</div>
@ -265,7 +282,7 @@ function ConfirmationBody({
onSubmit,
}: {
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
}) {
return (
<div className="flex flex-wrap items-center gap-2">
@ -273,15 +290,35 @@ function ConfirmationBody({
type="button"
disabled={submitting}
onClick={() => void onSubmit({ kind: "yes" })}
className={PRIMARY_BUTTON}
className={PRIMARY_BUTTON_CLASS}
>
{submitting ? <Spinner /> : <CheckIcon className="size-4" aria-hidden="true" />}
{submitting ? (
<Spinner className="size-4" />
) : (
<CheckIcon className="size-4" aria-hidden="true" />
)}
Confirm
</button>
</div>
);
}
/**
* Long labels wrap badly as pills, so they become a stacked list instead.
*/
export function shouldStackOptions(options: InterviewOption[]): boolean {
return options.some(
(option) =>
option.label.length > STACK_LABEL_LENGTH || Boolean(option.description),
);
}
function optionListClass(stacked: boolean): string {
return stacked
? "flex flex-col items-stretch gap-2"
: "flex flex-wrap items-center gap-2";
}
function ChoiceBody({
options,
allowFreeform,
@ -291,19 +328,23 @@ function ChoiceBody({
options: InterviewOption[];
allowFreeform: boolean;
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
}) {
const stacked = shouldStackOptions(options);
return (
<div className="space-y-4">
<div className="space-y-2.5">
{options.length > 0 && (
<div className="flex flex-wrap items-center gap-2">
<div className={optionListClass(stacked)}>
{options.map((option) => (
<button
key={option.key}
type="button"
disabled={submitting}
onClick={() => void onSubmit({ kind: "selected", option_key: option.key })}
className={CHOICE_BUTTON}
className={
stacked ? `${DOCK_CHOICE_BUTTON} justify-start` : DOCK_CHOICE_BUTTON
}
>
<OptionLabel option={option} />
</button>
@ -311,7 +352,7 @@ function ChoiceBody({
</div>
)}
{allowFreeform && (
<FreeformBody
<FreeformAnswer
submitting={submitting}
onSubmit={onSubmit}
placeholder={
@ -319,8 +360,6 @@ function ChoiceBody({
? "Or write a custom response…"
: "Write your response…"
}
submitLabel="Send"
divider={options.length > 0}
/>
)}
</div>
@ -334,7 +373,7 @@ function MultiSelectBody({
}: {
options: InterviewOption[];
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
}) {
const [selected, setSelected] = useState<Set<string>>(new Set());
@ -352,11 +391,14 @@ function MultiSelectBody({
if (selected.has(option.key)) selectedKeys.push(option.key);
}
const stacked = shouldStackOptions(options);
return (
<div className="space-y-3">
<div className="flex flex-wrap items-center gap-2">
<div className="space-y-2.5">
<div className={optionListClass(stacked)}>
{options.map((option) => {
const isSelected = selected.has(option.key);
const base = isSelected ? DOCK_CHOICE_BUTTON_SELECTED : DOCK_CHOICE_BUTTON;
return (
<button
key={option.key}
@ -364,7 +406,7 @@ function MultiSelectBody({
disabled={submitting}
aria-pressed={isSelected}
onClick={() => toggle(option.key)}
className={isSelected ? CHOICE_BUTTON_SELECTED : CHOICE_BUTTON}
className={stacked ? `${base} justify-start` : base}
>
{isSelected && <CheckIcon className="size-3.5" aria-hidden="true" />}
<OptionLabel option={option} />
@ -380,9 +422,13 @@ function MultiSelectBody({
type="button"
disabled={submitting || selectedKeys.length === 0}
onClick={() => void onSubmit({ kind: "multi_selected", option_keys: selectedKeys })}
className={PRIMARY_BUTTON}
className={PRIMARY_BUTTON_CLASS}
>
{submitting ? <Spinner /> : <CheckIcon className="size-4" aria-hidden="true" />}
{submitting ? (
<Spinner className="size-4" />
) : (
<CheckIcon className="size-4" aria-hidden="true" />
)}
Submit selection
</button>
</div>
@ -390,82 +436,23 @@ function MultiSelectBody({
);
}
function FreeformBody({
function FreeformAnswer({
submitting,
onSubmit,
placeholder,
submitLabel,
divider = false,
}: {
submitting: boolean;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<void>;
onSubmit: (answer: SubmitInterviewAnswer) => Promise<boolean>;
placeholder: string;
submitLabel: string;
divider?: boolean;
}) {
const [value, setValue] = useState("");
async function handleSubmit(event: FormEvent<HTMLFormElement>) {
event.preventDefault();
const trimmed = value.trim();
if (!trimmed || submitting) return;
await onSubmit({ kind: "text", text: trimmed });
setValue("");
}
function handleKeyDown(event: KeyboardEvent<HTMLTextAreaElement>) {
if (event.key === "Enter" && !event.shiftKey) {
event.preventDefault();
const form = event.currentTarget.form;
if (form) form.requestSubmit();
}
}
const disabled = submitting || value.trim().length === 0;
return (
<form onSubmit={handleSubmit} className="space-y-2">
{divider && (
<div className="flex items-center gap-3" aria-hidden="true">
<span className="h-px flex-1 bg-line" />
<span className="text-xs text-fg-muted">or</span>
<span className="h-px flex-1 bg-line" />
</div>
)}
<div className="flex items-end gap-2">
<label className="sr-only" htmlFor="interview-freeform-answer">
Your response
</label>
<textarea
id="interview-freeform-answer"
name="answer"
aria-label="Interview answer"
rows={1}
value={value}
onChange={(event) => setValue(event.target.value)}
onKeyDown={handleKeyDown}
placeholder={placeholder}
disabled={submitting}
className="block w-full resize-none rounded-lg bg-panel-alt px-3.5 py-2.5 text-base/6 text-fg outline-1 -outline-offset-1 outline-line-strong placeholder:text-fg-muted focus:outline-2 focus:-outline-offset-1 focus:outline-teal-500 disabled:opacity-60 sm:text-sm/5"
/>
<button type="submit" disabled={disabled} className={PRIMARY_BUTTON}>
{submitting ? (
<Spinner />
) : (
<ArrowUturnLeftIcon
className="size-3.5 -scale-x-100"
aria-hidden="true"
/>
)}
{submitLabel}
</button>
</div>
<p className="text-xs text-fg-muted">
Press <kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">Enter</kbd> to
send · <kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">Shift</kbd>+
<kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">Enter</kbd> for a new line
</p>
</form>
<DockComposer
onSubmit={(text) => onSubmit({ kind: "text", text })}
placeholder={placeholder}
submitLabel="Send"
submitting={submitting}
ariaLabel="Interview answer"
/>
);
}
@ -482,27 +469,6 @@ function OptionLabel({ option }: { option: InterviewOption }) {
);
}
function Spinner() {
return <ArrowPathIcon className="size-4 animate-spin" aria-hidden="true" />;
}
function questionTypeLabel(type: QuestionType): string {
switch (type) {
case QuestionType.YES_NO:
return "Yes or no";
case QuestionType.CONFIRMATION:
return "Confirmation required";
case QuestionType.MULTIPLE_CHOICE:
return "Pick one";
case QuestionType.MULTI_SELECT:
return "Pick one or more";
case QuestionType.FREEFORM:
return "Freeform response";
default:
return "";
}
}
function interviewSubmitErrorMessage(error: unknown): string {
if (error instanceof ApiError) {
return error.requestId

View file

@ -0,0 +1,59 @@
import { ArrowTopRightOnSquareIcon } from "@heroicons/react/20/solid";
import type { ReviewTarget } from "@qltysh/fabro-api-client";
/**
* Re-check the URL before putting it in an `href`. The server already rejects
* unsafe targets (see `ReviewTarget::new` in
* `lib/foundation/fabro-types/src/interview.rs`), but React does not sanitize
* `href`, so a `javascript:` URL reaching this component would execute. Length
* and control-character limits stay server-side; they cannot affect the DOM.
*/
export function safeReviewTarget(
target: ReviewTarget | null | undefined,
): ReviewTarget | null {
if (!target?.url || !target.label) return null;
try {
const parsed = new URL(target.url);
const safe =
(parsed.protocol === "http:" || parsed.protocol === "https:") &&
Boolean(parsed.host) &&
!parsed.username &&
!parsed.password;
return safe ? target : null;
} catch {
return null;
}
}
/**
* The review question sentence, with the target label as an external link.
* Mirrors `ReviewTarget::question_text_with_link` in
* `lib/foundation/fabro-types/src/interview.rs`.
*/
export function ReviewTargetQuestion({
target,
className,
}: {
target: ReviewTarget;
className?: string;
}) {
return (
<p className={className}>
Review the{" "}
<a
href={target.url}
target="_blank"
rel="noopener noreferrer"
referrerPolicy="no-referrer"
className="inline-flex items-baseline gap-1 font-semibold text-teal-300 underline decoration-teal-500/50 underline-offset-2 transition-colors hover:text-fg focus-visible:rounded-sm focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500"
>
<span>{target.label}</span>
<ArrowTopRightOnSquareIcon
className="size-3 shrink-0 self-center"
aria-hidden="true"
/>
</a>{" "}
{target.kind}, then choose the next action.
</p>
);
}

View file

@ -0,0 +1,118 @@
import {
afterEach,
beforeEach,
describe,
expect,
mock,
test,
} from "bun:test";
import TestRenderer, { act } from "react-test-renderer";
import { setupReactTestEnv } from "../lib/test-utils";
import { DockComposer } from "./run-dock";
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
let teardownReactEnv: (() => void) | undefined;
beforeEach(() => {
teardownReactEnv = setupReactTestEnv();
});
afterEach(() => {
for (const renderer of mountedRenderers.splice(0)) {
act(() => renderer.unmount());
}
teardownReactEnv?.();
teardownReactEnv = undefined;
});
describe("DockComposer", () => {
test("describes its keyboard behavior to assistive technology", () => {
let renderer!: TestRenderer.ReactTestRenderer;
act(() => {
renderer = TestRenderer.create(
<DockComposer
onSubmit={() => Promise.resolve(true)}
placeholder="Write a message"
submitLabel="Send"
submitting={false}
ariaLabel="Message"
/>,
);
});
mountedRenderers.push(renderer);
const textarea = renderer.root.findByType("textarea");
const instruction = renderer.root.findByProps({
id: textarea.props["aria-describedby"],
});
expect(instruction.children.join("")).toBe(
"Press Enter to send. Press Shift+Enter for a new line.",
);
expect(textarea.props.name).toBeUndefined();
});
test("does not submit Enter while an IME composition is active", async () => {
const onSubmit = mock(() => Promise.resolve(true));
const preventDefault = mock(() => undefined);
let renderer!: TestRenderer.ReactTestRenderer;
act(() => {
renderer = TestRenderer.create(
<DockComposer
onSubmit={onSubmit}
placeholder="Write a message"
submitLabel="Send"
submitting={false}
ariaLabel="Message"
/>,
);
});
mountedRenderers.push(renderer);
const textarea = renderer.root.findByType("textarea");
act(() => textarea.props.onChange({ target: { value: "draft" } }));
await act(async () => {
textarea.props.onKeyDown({
key: "Enter",
shiftKey: false,
nativeEvent: { isComposing: true },
preventDefault,
});
});
expect(preventDefault).not.toHaveBeenCalled();
expect(onSubmit).not.toHaveBeenCalled();
});
test("submits Enter after composition ends", async () => {
const onSubmit = mock(() => Promise.resolve(true));
const preventDefault = mock(() => undefined);
let renderer!: TestRenderer.ReactTestRenderer;
act(() => {
renderer = TestRenderer.create(
<DockComposer
onSubmit={onSubmit}
placeholder="Write a message"
submitLabel="Send"
submitting={false}
ariaLabel="Message"
/>,
);
});
mountedRenderers.push(renderer);
const textarea = renderer.root.findByType("textarea");
act(() => textarea.props.onChange({ target: { value: " ready " } }));
await act(async () => {
textarea.props.onKeyDown({
key: "Enter",
shiftKey: false,
nativeEvent: { isComposing: false },
preventDefault,
});
});
expect(preventDefault).toHaveBeenCalledTimes(1);
expect(onSubmit).toHaveBeenCalledWith("ready");
});
});

View file

@ -0,0 +1,317 @@
import {
useId,
useState,
type FormEvent,
type KeyboardEvent,
type ReactNode,
type Ref,
} from "react";
import {
ArrowUturnLeftIcon,
ChevronUpIcon,
} from "@heroicons/react/20/solid";
import { classNames } from "../lib/class-names";
import { Spinner } from "./state";
import {
INPUT_CLASS,
PRIMARY_BUTTON_CLASS,
} from "./ui";
/**
* Shared chrome for the two controls docked at the bottom of the run detail
* route: the interview question panel and the steering composer.
*
* The shell is three zones. The header is always visible and doubles as the
* collapsed bar. The body scrolls. The actions stay pinned, so the controls
* needed to answer or send never scroll out of reach.
*
* Collapsed state is owned by the caller. Each dock has its own rule for when
* a collapsed panel must reopen — a new question for the interview, a run
* waiting for steering for the composer — and those rules are clearer next to
* the state they depend on.
*/
/** Ceiling on the expanded dock, so a long body cannot take the page. */
const DOCK_MAX_HEIGHT = "max-h-[60vh]";
export const DOCK_HEADER_BUTTON =
"inline-flex shrink-0 items-center gap-1.5 rounded-md bg-overlay px-2 py-1 text-xs font-medium text-fg-2 outline-1 -outline-offset-1 outline-line-strong transition-colors hover:bg-overlay-strong hover:text-fg focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500 disabled:cursor-not-allowed disabled:opacity-50 disabled:hover:bg-overlay disabled:hover:text-fg-2";
export const DOCK_CHOICE_BUTTON =
"inline-flex items-center justify-center gap-1.5 rounded-lg bg-overlay px-3.5 py-2 text-left text-sm font-medium text-fg-2 outline-1 -outline-offset-1 outline-line-strong transition-colors hover:bg-overlay-strong hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500 disabled:cursor-not-allowed disabled:opacity-60";
export const DOCK_CHOICE_BUTTON_SELECTED =
"inline-flex items-center justify-center gap-1.5 rounded-lg bg-teal-500/15 px-3.5 py-2 text-left text-sm font-medium text-fg outline-1 -outline-offset-1 outline-teal-500/60 transition-colors hover:bg-teal-500/20 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500";
/**
* How the dock signals its state.
*
* - `waiting` pulses amber: the run is blocked on the operator, as expected.
* - `alert` pulses amber and colors the label too, for a run that went off
* its normal path and is stuck until someone acts.
* - `idle` is a resting control with no pending demand.
*/
export type DockTone = "waiting" | "alert" | "idle";
export interface RunDockShellProps {
/** Accessible name for the docked region. */
label: string;
className?: string;
tone: DockTone;
/** Short state phrase, e.g. "Awaiting input". */
status: string;
/** Stage name shown in mono beside the status. */
stage?: string | null;
/** One-line summary shown only while collapsed. */
peek?: string | null;
/** Run-level controls, rendered beside the collapse toggle. */
headerActions?: ReactNode;
/** Scrolling zone. Omit when there is nothing to scroll. */
body?: ReactNode;
/** Pinned zone. */
actions: ReactNode;
collapsed: boolean;
onCollapsedChange: (collapsed: boolean) => void;
}
export function RunDockShell({
label,
className,
tone,
status,
stage,
peek,
headerActions,
body,
actions,
collapsed,
onCollapsedChange,
}: RunDockShellProps) {
const contentId = useId();
return (
<section
aria-label={label}
className={classNames(
"flex flex-col",
!collapsed && DOCK_MAX_HEIGHT,
className,
)}
>
<div className="flex shrink-0 items-center gap-2.5 px-5 py-2 sm:px-6">
<StatusDot tone={tone} />
{/* A live region: the dock changing to a state that needs the
operator has to reach assistive tech, not only the eye. */}
<span
role="status"
className={`shrink-0 text-sm font-medium ${
tone === "alert" ? "text-amber" : "text-fg-2"
}`}
>
{status}
</span>
{stage && (
<>
<span className="shrink-0 text-fg-muted" aria-hidden="true">
·
</span>
<span className="min-w-0 truncate font-mono text-xs text-fg-3">
{stage}
</span>
</>
)}
{collapsed && peek ? (
<span className="min-w-0 flex-1 truncate text-sm text-fg-3">
· {peek}
</span>
) : (
<span className="flex-1" />
)}
{headerActions}
<button
type="button"
onClick={() => onCollapsedChange(!collapsed)}
aria-expanded={!collapsed}
aria-controls={contentId}
aria-label={collapsed ? `Expand ${label}` : `Collapse ${label}`}
className="inline-flex size-6.5 shrink-0 items-center justify-center rounded-md text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500"
>
<ChevronUpIcon
className={`size-4 transition-transform duration-200 ease-[cubic-bezier(0.16,1,0.3,1)] ${
collapsed ? "rotate-180" : ""
}`}
aria-hidden="true"
/>
</button>
</div>
{/* Hidden rather than unmounted, so a half-written message survives a
collapse. The display utility is swapped rather than layered, so two
display classes cannot collide in the cascade. */}
<div
id={contentId}
className={collapsed ? "hidden" : "flex min-h-0 flex-1 flex-col"}
>
{body && (
<div className="min-h-0 flex-1 space-y-3 overflow-y-auto border-t border-line px-5 pt-3.5 pb-1 sm:px-6">
{body}
</div>
)}
<div
className={classNames(
"flex max-h-[50%] shrink-0 flex-col gap-2.5 overflow-y-auto px-5 pt-2.5 pb-3.5 sm:px-6",
!body && "border-t border-line",
)}
>
{actions}
</div>
</div>
</section>
);
}
function StatusDot({ tone }: { tone: DockTone }) {
if (tone === "idle") {
return (
<span
className="size-2 shrink-0 rounded-full bg-fg-muted"
aria-hidden="true"
/>
);
}
return (
<span
className="relative flex size-2 shrink-0 items-center justify-center"
aria-hidden="true"
>
<span className="absolute inline-flex size-full animate-ping rounded-full bg-amber/60" />
<span className="relative inline-flex size-2 rounded-full bg-amber" />
</span>
);
}
export interface DockComposerProps {
/**
* Sends the trimmed text. Resolve `true` to clear the box; resolve `false`
* to keep what the operator typed, so a failed send is not lost.
*/
onSubmit: (text: string) => Promise<boolean>;
placeholder: string;
submitLabel: string;
pendingLabel?: string;
submitting: boolean;
disabled?: boolean;
ariaLabel: string;
className?: string;
maxLength?: number;
textareaRef?: Ref<HTMLTextAreaElement>;
}
/**
* The single composer used by both docks: the interview's freeform answer and
* the steering message. Enter sends, Shift+Enter breaks the line, and the
* hint for that only appears on focus — inside the row, so revealing it does
* not shift the layout.
*/
export function DockComposer({
onSubmit,
placeholder,
submitLabel,
pendingLabel,
submitting,
disabled = false,
ariaLabel,
className,
maxLength,
textareaRef,
}: DockComposerProps) {
const [value, setValue] = useState("");
const fieldId = useId();
const instructionId = `${fieldId}-instruction`;
const trimmed = value.trim();
const composerDisabled = disabled || submitting;
const canSend = trimmed.length > 0 && !composerDisabled;
async function send() {
if (!canSend) return;
const cleared = await onSubmit(trimmed);
if (cleared) setValue("");
}
function handleSubmit(event: FormEvent<HTMLFormElement>) {
event.preventDefault();
void send();
}
function handleKeyDown(event: KeyboardEvent<HTMLTextAreaElement>) {
if (
event.key === "Enter" &&
!event.shiftKey &&
!event.nativeEvent.isComposing
) {
event.preventDefault();
void send();
}
}
return (
<form
onSubmit={handleSubmit}
className={classNames("group flex items-end gap-2", className)}
>
<label className="sr-only" htmlFor={fieldId}>
{ariaLabel}
</label>
<p id={instructionId} className="sr-only">
Press Enter to send. Press Shift+Enter for a new line.
</p>
<textarea
id={fieldId}
ref={textareaRef}
aria-label={ariaLabel}
aria-describedby={instructionId}
rows={1}
value={value}
maxLength={maxLength}
onChange={(event) => setValue(event.target.value)}
onKeyDown={handleKeyDown}
placeholder={placeholder}
disabled={composerDisabled}
className={`${INPUT_CLASS} min-w-0 flex-1 resize-none disabled:opacity-60`}
/>
<p
aria-hidden="true"
className="pointer-events-none hidden shrink-0 items-center gap-1 pb-2.5 text-xs whitespace-nowrap text-fg-muted opacity-0 transition-opacity group-focus-within:opacity-100 md:flex"
>
<kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">
Enter
</kbd>
send
<kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">
Shift
</kbd>
+
<kbd className="rounded bg-overlay px-1 font-mono text-[0.6875rem]">
Enter
</kbd>
newline
</p>
<button
type="submit"
disabled={!canSend}
className={PRIMARY_BUTTON_CLASS}
>
{submitting ? (
<Spinner className="size-4" />
) : (
<ArrowUturnLeftIcon
className="size-3.5 -scale-x-100"
aria-hidden="true"
/>
)}
{submitting && pendingLabel ? pendingLabel : submitLabel}
</button>
</form>
);
}

View file

@ -115,8 +115,10 @@ export function RunTableRow({
</td>
)}
{show("size") && (
<td className="whitespace-nowrap px-3 py-2.5 text-center">
{run.size != null && <SizeChip size={run.size} />}
<td className="relative z-10 px-3 py-2.5 text-center whitespace-nowrap">
{run.size != null && (
<SizeChip size={run.size} totalUsdMicros={run.totalUsdMicros} />
)}
</td>
)}
{show("changes") && (

View file

@ -0,0 +1,58 @@
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import TestRenderer, { act } from "react-test-renderer";
import { setupReactTestEnv } from "../lib/test-utils";
import { SizeChip } from "./size-chip";
import { Tooltip } from "./ui";
let teardownReactTestEnv: (() => void) | undefined;
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
function render(element: React.ReactElement): TestRenderer.ReactTestRenderer {
let renderer: TestRenderer.ReactTestRenderer | undefined;
act(() => {
renderer = TestRenderer.create(element);
});
mountedRenderers.push(renderer!);
return renderer!;
}
function tooltipLabel(element: React.ReactElement): string {
return render(element).root.findByType(Tooltip).props.label as string;
}
describe("SizeChip", () => {
beforeEach(() => {
teardownReactTestEnv = setupReactTestEnv();
});
afterEach(() => {
act(() => {
for (const renderer of mountedRenderers.splice(0)) {
renderer.unmount();
}
});
teardownReactTestEnv?.();
teardownReactTestEnv = undefined;
});
test("renders the size letter", () => {
expect(JSON.stringify(render(<SizeChip size="M" />).toJSON())).toContain("M");
});
test("appends the cost to the tooltip", () => {
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={12_340_000} />))
.toBe("Size M · $12.34");
});
test("omits the cost when the run has no billing yet", () => {
expect(tooltipLabel(<SizeChip size="M" />)).toBe("Size M");
expect(tooltipLabel(<SizeChip size="M" totalUsdMicros={null} />)).toBe("Size M");
});
test("calls out the tiers that warrant attention", () => {
expect(tooltipLabel(<SizeChip size="L" totalUsdMicros={150_000_000} />))
.toBe("Size L (risky) · $150.00");
expect(tooltipLabel(<SizeChip size="XL" />)).toBe("Size XL (unhealthy)");
});
});

View file

@ -1,3 +1,4 @@
import { memo } from "react";
import type { RunSize } from "@qltysh/fabro-api-client";
import { formatUsdMicros } from "../lib/format";
@ -11,7 +12,7 @@ const SIZE_TONE: Record<RunSize, { className: string; note: string | null }> = {
XL: { className: "bg-coral/15 text-coral", note: "unhealthy" },
};
export function SizeChip({
export const SizeChip = memo(function SizeChip({
size,
totalUsdMicros,
}: {
@ -19,10 +20,10 @@ export function SizeChip({
totalUsdMicros?: number | null;
}) {
const tone = SIZE_TONE[size];
const billed = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)} billed` : "";
const amount = totalUsdMicros != null ? ` · ${formatUsdMicros(totalUsdMicros)}` : "";
const tooltip = tone.note != null
? `Size ${size} (${tone.note})${billed}`
: `Size ${size}${billed}`;
? `Size ${size} (${tone.note})${amount}`
: `Size ${size}${amount}`;
return (
<Tooltip label={tooltip}>
<span className={`rounded px-1.5 py-0.5 font-mono text-xs font-bold tabular-nums ${tone.className}`}>
@ -30,4 +31,4 @@ export function SizeChip({
</span>
</Tooltip>
);
}
});

View file

@ -9,6 +9,7 @@ import { StagePopover } from "./stage-popover";
import { deriveStageSummary } from "./stage-popover-summary";
import type { Stage } from "../lib/stage-sidebar";
import { generatedAxios } from "../lib/api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
function makeEvent(overrides: Partial<EventEnvelope>): EventEnvelope {
return {
@ -34,6 +35,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
duration: "1m 30s",
startedAt: "2026-05-24T11:58:30Z",
providerUsed: { mode: "policy", model: "claude-opus-4-7", reasoning_effort: "high" },
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -3,6 +3,7 @@ import type { EventEnvelope } from "@qltysh/fabro-api-client";
import TestRenderer, { act } from "react-test-renderer";
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
import type { Stage } from "../stage-sidebar";
import { FanInResults } from "./fan-in-results";
@ -22,6 +23,7 @@ const fanInStage: Stage = {
visit: 1,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
};
function event(seq: number, partial: Partial<EventEnvelope>): EventEnvelope {

View file

@ -98,6 +98,33 @@ describe("parseHumanInterviewPairs", () => {
});
});
test("preserves a typed review target from started events", () => {
const events: EventEnvelope[] = [
makeEventEnvelope(1, {
event: "interview.started",
properties: {
question_id: "q-1",
question:
"Review the Quarry review exercise document, then choose the next action.",
question_type: "multiple_choice",
review_target: {
label: "Quarry review exercise",
url: "https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
kind: "document",
},
},
}),
];
const pairs = parseHumanInterviewPairs(events);
expect(pairs[0].question.reviewTarget).toEqual({
label: "Quarry review exercise",
url: "https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
kind: "document",
});
});
test("captures timeout and interrupted resolutions", () => {
const events: EventEnvelope[] = [
makeEventEnvelope(1, {

View file

@ -1,7 +1,14 @@
import { StageOutcome } from "@qltysh/fabro-api-client";
import type { EventEnvelope } from "@qltysh/fabro-api-client";
import { ReviewTargetKind, StageOutcome } from "@qltysh/fabro-api-client";
import type { EventEnvelope, ReviewTarget } from "@qltysh/fabro-api-client";
import { getArray, getNumber, getObject, getString, type UnknownRecord } from "../../lib/unknown";
import {
getArray,
getNumber,
getObject,
getString,
isRecord,
type UnknownRecord,
} from "../../lib/unknown";
const STAGE_OUTCOMES: ReadonlySet<string> = new Set(Object.values(StageOutcome));
@ -25,6 +32,7 @@ export interface HumanQuestion {
allowFreeform: boolean;
timeoutSeconds: number | null;
contextDisplay: string | null;
reviewTarget: ReviewTarget | null;
}
export type HumanResolution =
@ -75,6 +83,15 @@ function parseInterviewOptions(value: unknown): InterviewOption[] {
return out;
}
function parseReviewTarget(value: unknown): ReviewTarget | null {
if (!isRecord(value)) return null;
const label = getString(value, "label");
const url = getString(value, "url");
const kind = getString(value, "kind");
if (!label || !url || kind !== ReviewTargetKind.DOCUMENT) return null;
return { label, url, kind };
}
/**
* Pair `interview.started` events with the matching `interview.completed`,
* `.timeout`, or `.interrupted` resolution by `question_id`. Unanswered
@ -98,6 +115,7 @@ export function parseHumanInterviewPairs(events: EventEnvelope[]): HumanIntervie
allowFreeform: props.allow_freeform === true,
timeoutSeconds: getNumber(props, "timeout_seconds") ?? null,
contextDisplay: getString(props, "context_display") ?? null,
reviewTarget: parseReviewTarget(props.review_target),
},
resolution: null,
});

View file

@ -10,6 +10,10 @@ import {
import type { EventEnvelope } from "@qltysh/fabro-api-client";
import type { Stage } from "../stage-sidebar";
import {
ReviewTargetQuestion,
safeReviewTarget,
} from "../review-target-question";
import { Tooltip } from "../ui";
import { formatAbsoluteTs, formatDurationMs } from "../../lib/format";
import { ACTIVE_STAGE_STATES } from "../../lib/stage-sidebar";
@ -148,6 +152,7 @@ function QuestionBlock({
stageActive: boolean;
}) {
const { question, resolution } = pair;
const reviewTarget = safeReviewTarget(question.reviewTarget);
return (
<article className="space-y-3">
<header className="flex flex-wrap items-baseline gap-x-3 gap-y-1">
@ -168,7 +173,14 @@ function QuestionBlock({
</header>
<div className="rounded-lg bg-panel p-4 outline-1 -outline-offset-1 outline-line">
<Markdown content={question.question} />
{reviewTarget ? (
<ReviewTargetQuestion
target={reviewTarget}
className="text-sm/6 text-fg-2"
/>
) : (
<Markdown content={question.question} />
)}
{question.contextDisplay && (
<div className="mt-3 border-t border-line pt-3 text-xs text-fg-muted">
<Markdown content={question.contextDisplay} />

View file

@ -4,6 +4,7 @@ import TestRenderer, { act } from "react-test-renderer";
import { MemoryRouter } from "react-router";
import { makeEventEnvelope, setupReactTestEnv } from "../../lib/test-utils";
import { makeBilledTokenCounts } from "../../lib/test-fixtures";
import type { Stage } from "../stage-sidebar";
import { ParallelChildren } from "./parallel-children";
@ -23,6 +24,7 @@ const parallelStage: Stage = {
visit: 1,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
};
function event(partial: Partial<EventEnvelope>): EventEnvelope {

View file

@ -2,6 +2,7 @@ import { describe, expect, test } from "bun:test";
import TestRenderer, { act } from "react-test-renderer";
import { MemoryRouter } from "react-router";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { StageSidebar, type Stage } from "./stage-sidebar";
function makeStage(overrides: Partial<Stage> = {}): Stage {
@ -17,6 +18,7 @@ function makeStage(overrides: Partial<Stage> = {}): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -1,22 +1,146 @@
import { describe, expect, test } from "bun:test";
import { createElement } from "react";
import { renderToStaticMarkup } from "react-dom/server";
import {
afterEach,
beforeEach,
describe,
expect,
mock,
test,
} from "bun:test";
import { createRef } from "react";
import TestRenderer, { act } from "react-test-renderer";
import { setupReactTestEnv } from "../lib/test-utils";
let steerPending = false;
let interruptPending = false;
const steerTrigger = mock(() => Promise.resolve(undefined));
const interruptTrigger = mock(() => Promise.resolve(undefined));
mock.module("../lib/mutations", () => ({
useSteerRun: () => ({
isMutating: steerPending,
trigger: steerTrigger,
}),
useInterruptRun: () => ({
isMutating: interruptPending,
trigger: interruptTrigger,
}),
}));
const {
isInterruptDisabled,
SteerWaitingStatus,
} from "./steer-bar";
isSteerDockCollapsed,
SteerBar,
steerStatusLabel,
} = await import("./steer-bar");
type SteerBarHandle = import("./steer-bar").SteerBarHandle;
mock.restore();
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
let teardownReactEnv: (() => void) | undefined;
function textFromNode(node: TestRenderer.ReactTestInstance): string {
return node.children
.map((child) =>
typeof child === "string" ? child : textFromNode(child),
)
.join("");
}
beforeEach(() => {
teardownReactEnv = setupReactTestEnv();
steerPending = false;
interruptPending = false;
steerTrigger.mockClear();
interruptTrigger.mockClear();
});
afterEach(() => {
for (const renderer of mountedRenderers.splice(0)) {
act(() => renderer.unmount());
}
teardownReactEnv?.();
teardownReactEnv = undefined;
});
describe("SteerBar", () => {
test("shows durable waiting state and prevents a second interrupt", () => {
test("prevents a second interrupt while one is in flight or already settled", () => {
expect(isInterruptDisabled(true, false)).toBe(true);
expect(isInterruptDisabled(false, true)).toBe(true);
expect(isInterruptDisabled(false, false)).toBe(false);
});
const html = renderToStaticMarkup(
createElement(SteerWaitingStatus, { waitingForSteer: true }),
test("names the durable waiting state in the dock header", () => {
expect(steerStatusLabel(true)).toBe("Interrupted — waiting for steering");
expect(steerStatusLabel(false)).toBe("Steering");
});
test("reopens the dock while the run waits for steering", () => {
expect(isSteerDockCollapsed(true, false)).toBe(true);
expect(isSteerDockCollapsed(false, false)).toBe(false);
// Collapsing cannot hide a run that is blocked on the operator.
expect(isSteerDockCollapsed(true, true)).toBe(false);
expect(isSteerDockCollapsed(false, true)).toBe(false);
});
test("the focus handle expands a collapsed dock before focusing", async () => {
const focus = mock(() => undefined);
const ref = createRef<SteerBarHandle>();
let renderer!: TestRenderer.ReactTestRenderer;
await act(async () => {
renderer = TestRenderer.create(
<SteerBar ref={ref} runId="run-1" />,
{
createNodeMock: (element) =>
element.type === "textarea" ? { focus } : null,
},
);
});
mountedRenderers.push(renderer);
act(() => {
renderer.root
.findByProps({ "aria-label": "Collapse Steer running agent" })
.props.onClick();
});
expect(
renderer.root.findByProps({
"aria-label": "Expand Steer running agent",
}),
).toBeDefined();
act(() => ref.current?.focus());
expect(
renderer.root.findByProps({
"aria-label": "Collapse Steer running agent",
}),
).toBeDefined();
expect(focus).not.toHaveBeenCalled();
await act(
async () =>
await new Promise((resolve) => {
setTimeout(resolve, 0);
}),
);
expect(html).toContain('role="status"');
expect(html).toContain("Interrupted — waiting for steering");
expect(focus).toHaveBeenCalledTimes(1);
});
test("interrupt progress disables the composer without calling it sending", async () => {
interruptPending = true;
let renderer!: TestRenderer.ReactTestRenderer;
await act(async () => {
renderer = TestRenderer.create(<SteerBar runId="run-1" />);
});
mountedRenderers.push(renderer);
const submit = renderer.root.findByProps({ type: "submit" });
expect(submit.props.disabled).toBe(true);
expect(textFromNode(submit)).toBe("Send");
expect(
renderer.root
.findAllByType("button")
.map(textFromNode),
).toContain("Interrupting…");
});
});

View file

@ -2,15 +2,22 @@ import {
useImperativeHandle,
useRef,
useState,
type FormEvent,
type KeyboardEvent,
type Ref,
} from "react";
import { StopIcon } from "@heroicons/react/20/solid";
import { ApiError } from "../lib/api-client";
import { classNames } from "../lib/class-names";
import { useInterruptRun, useSteerRun } from "../lib/mutations";
import {
DockComposer,
RunDockShell,
DOCK_HEADER_BUTTON,
} from "./run-dock";
import { ErrorMessage } from "./ui";
const STEER_MAX_LENGTH = 8192;
export interface SteerBarProps {
runId: string;
waitingForSteer?: boolean;
@ -28,17 +35,20 @@ export function isInterruptDisabled(
return waitingForSteer || mutationPending;
}
export function SteerWaitingStatus({
waitingForSteer,
}: {
waitingForSteer: boolean;
}) {
if (!waitingForSteer) return null;
return (
<p role="status" className="mt-2 text-xs text-amber">
Interrupted — waiting for steering
</p>
);
/**
* A run that is waiting for steering needs the operator, so the dock reopens
* itself and stays open until the wait clears. Derived rather than stored, so
* collapsing during the wait cannot hide the prompt.
*/
export function isSteerDockCollapsed(
collapsePreferred: boolean,
waitingForSteer: boolean,
): boolean {
return collapsePreferred && !waitingForSteer;
}
export function steerStatusLabel(waitingForSteer: boolean): string {
return waitingForSteer ? "Interrupted — waiting for steering" : "Steering";
}
export function SteerBar({
@ -46,31 +56,38 @@ export function SteerBar({
waitingForSteer = false,
ref,
}: SteerBarProps) {
const [text, setText] = useState("");
const [errorMessage, setErrorMessage] = useState<string | null>(null);
const [collapsePreferred, setCollapsePreferred] = useState(false);
const textareaRef = useRef<HTMLTextAreaElement | null>(null);
const steer = useSteerRun(runId);
const interrupt = useInterruptRun(runId);
const pending = steer.isMutating || interrupt.isMutating;
const interruptDisabled = isInterruptDisabled(waitingForSteer, pending);
const collapsed = isSteerDockCollapsed(collapsePreferred, waitingForSteer);
useImperativeHandle(ref, () => ({
focus() {
textareaRef.current?.focus();
},
}));
useImperativeHandle(
ref,
() => ({
focus() {
if (!collapsed) {
textareaRef.current?.focus();
return;
}
setCollapsePreferred(false);
setTimeout(() => textareaRef.current?.focus(), 0);
},
}),
[collapsed],
);
const trimmed = text.trim();
const canSend = trimmed.length > 0 && !pending;
async function sendSteering() {
if (!canSend) return;
async function sendSteering(text: string) {
setErrorMessage(null);
try {
await steer.trigger({ text: trimmed, interrupt: false });
setText("");
await steer.trigger({ text, interrupt: false });
return true;
} catch (err) {
setErrorMessage(formatSteerError(err));
return false;
}
}
@ -84,59 +101,47 @@ export function SteerBar({
}
}
function handleSubmit(e: FormEvent) {
e.preventDefault();
void sendSteering();
}
function handleKeyDown(e: KeyboardEvent<HTMLTextAreaElement>) {
if (e.key === "Enter" && !e.shiftKey) {
e.preventDefault();
void sendSteering();
}
}
return (
<form
onSubmit={handleSubmit}
aria-label="Steer running agent"
className="mx-auto max-w-4xl px-4 py-3 sm:px-6 lg:px-8"
>
<div className="flex items-end gap-2">
<textarea
ref={textareaRef}
value={text}
onChange={(e) => setText(e.target.value)}
onKeyDown={handleKeyDown}
placeholder="Steer the agent…"
rows={1}
maxLength={8192}
aria-label="Steering message"
className="flex-1 resize-none rounded-md bg-overlay px-3 py-2 text-sm text-fg outline-1 -outline-offset-1 outline-line-strong placeholder:text-fg-muted focus:outline-2 focus:-outline-offset-1 focus:outline-teal-500"
/>
<RunDockShell
label="Steer running agent"
tone={waitingForSteer ? "alert" : "idle"}
status={steerStatusLabel(waitingForSteer)}
peek="Send a message to the running agent"
collapsed={collapsed}
onCollapsedChange={setCollapsePreferred}
headerActions={
// Interrupt acts on the run, not on the message being composed, so it
// sits with the other run-level controls instead of in the composer.
<button
type="button"
onClick={() => void fireInterrupt()}
disabled={interruptDisabled}
className="inline-flex shrink-0 items-center gap-2 rounded-md bg-overlay px-3 py-2 text-sm font-medium text-amber outline-1 -outline-offset-1 outline-amber/40 transition-colors hover:bg-amber/15 focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-amber disabled:cursor-not-allowed disabled:opacity-60"
className={classNames(
DOCK_HEADER_BUTTON,
"text-amber outline-amber/40 hover:bg-amber/15 hover:text-amber focus-visible:outline-amber disabled:hover:bg-overlay disabled:hover:text-amber",
)}
>
<StopIcon className="size-3" aria-hidden="true" />
{interrupt.isMutating ? "Interrupting…" : "Interrupt"}
</button>
<button
type="submit"
disabled={!canSend}
className="inline-flex shrink-0 items-center justify-center rounded-md bg-teal-500 px-4 py-2 text-sm font-medium text-on-primary transition-colors hover:bg-teal-300 focus-visible:outline-2 focus-visible:outline-offset-2 focus-visible:outline-teal-500 disabled:cursor-not-allowed disabled:opacity-60 disabled:hover:bg-teal-500"
>
{steer.isMutating ? "Sending…" : "Send"}
</button>
</div>
{errorMessage && (
<div className="mt-2">
<ErrorMessage message={errorMessage} />
</div>
)}
<SteerWaitingStatus waitingForSteer={waitingForSteer} />
</form>
}
actions={
<>
<DockComposer
onSubmit={sendSteering}
placeholder="Steer the agent…"
submitLabel="Send"
pendingLabel="Sending…"
submitting={steer.isMutating}
disabled={pending}
ariaLabel="Steering message"
maxLength={STEER_MAX_LENGTH}
textareaRef={textareaRef}
/>
{errorMessage && <ErrorMessage message={errorMessage} />}
</>
}
/>
);
}

View file

@ -103,6 +103,17 @@ describe("mapRunListItem", () => {
expect(mapRunListItem(summary).title).toBe("Untitled run");
});
test("carries the billed total so the size chip can show it on hover", () => {
expect(mapRunListItem(makeRun()).totalUsdMicros).toBe(500000);
});
test("leaves the billed total undefined for runs without terminal billing", () => {
expect(mapRunListItem(makeRun({ billing: null })).totalUsdMicros).toBeUndefined();
expect(
mapRunListItem(makeRun({ billing: { total_usd_micros: null } })).totalUsdMicros,
).toBeUndefined();
});
});
describe("mapRunToRunItem", () => {

View file

@ -46,6 +46,7 @@ export interface RunItem {
createdBy: Principal;
lastEventAt?: string;
size?: RunSize;
totalUsdMicros?: number;
}
export const columnStatuses = [
@ -119,6 +120,7 @@ export function mapRunListItem(item: Run): RunItem {
additions: item.diff?.additions,
deletions: item.diff?.deletions,
size: item.size,
totalUsdMicros: item.billing?.total_usd_micros ?? undefined,
};
}

View file

@ -0,0 +1,28 @@
import type { BilledTokenCounts } from "@qltysh/fabro-api-client";
export interface BillingTokenBucket {
label: string;
value: number;
}
export function billableOutputTokens(billing: BilledTokenCounts): number {
return billing.output_tokens + billing.reasoning_tokens;
}
/** The disjoint token buckets shown in every billing breakdown. */
export function billingTokenBuckets(billing: BilledTokenCounts): BillingTokenBucket[] {
return [
{ label: "Cache read", value: billing.cache_read_tokens },
{ label: "Cache creation", value: billing.cache_write_tokens },
{ label: "Uncached", value: billing.input_tokens },
{ label: "Output", value: billableOutputTokens(billing) },
];
}
export function hasBillingUsage(billing: BilledTokenCounts): boolean {
return (
billing.total_tokens !== 0 ||
(billing.total_usd_micros ?? 0) !== 0 ||
billingTokenBuckets(billing).some((bucket) => bucket.value !== 0)
);
}

View file

@ -0,0 +1,5 @@
export function classNames(
...classes: Array<string | false | null | undefined>
): string {
return classes.filter(Boolean).join(" ");
}

View file

@ -1,3 +1,4 @@
import { useCallback } from "react";
import useSWR, { type SWRConfiguration } from "swr";
import type {
ApiQuestion,
@ -73,6 +74,7 @@ import {
type RunFileSelection,
type RunGraphDirection,
} from "./query-keys";
import { isTerminalRunStatus } from "./run-actions";
const immutableOptions: SWRConfiguration = {
revalidateIfStale: false,
@ -179,10 +181,21 @@ export function useRunsPage(opts: RunsPageOptions = {}, enabled = true) {
);
}
export function useRun(id: string | undefined) {
export function useRun(id: string | undefined, refreshInterval?: number) {
const pollingInterval = useCallback(
(run: Run | null | undefined) =>
refreshInterval &&
run?.timestamps.started_at &&
!isTerminalRunStatus(run.lifecycle.status.kind)
? refreshInterval
: 0,
[refreshInterval],
);
return useSWR<Run | null>(
id ? queryKeys.runs.detail(id) : null,
() => apiNullableData(() => runsApi.retrieveRun(id!)),
refreshInterval ? { refreshInterval: pollingInterval } : undefined,
);
}

View file

@ -1,7 +1,6 @@
import { describe, expect, test } from "bun:test";
import { queryKeys } from "./query-keys";
import { queryKeysForRunEvent } from "./run-events";
describe("queryKeys", () => {
test("uses semantic tuples as stable SWR keys and keeps SSE URLs explicit", () => {
@ -61,52 +60,4 @@ describe("queryKeys", () => {
expect(queryKeys.runs.attachUrl("run 1")).toBe("/api/v1/runs/run%201/attach");
});
test("event-mapped keys match query hook resources", () => {
expect(queryKeysForRunEvent("run-1", "checkpoint.completed")).toEqual(
[
...queryKeys.runs.filesAllScopes("run-1"),
queryKeys.runs.commits("run-1"),
],
);
expect(queryKeysForRunEvent("run-1", "stage.completed", "stage-1")).toEqual([
queryKeys.runs.stages("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.graph("run-1", "LR"),
queryKeys.runs.graph("run-1", "TB"),
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
expect(queryKeysForRunEvent("run-1", "run.title.updated")).toEqual([
queryKeys.runs.detail("run-1"),
]);
});
test("agent activity events invalidate per-stage resources", () => {
for (const event of [
"stage.prompt",
"agent.tool.started",
"agent.tool.completed",
"command.started",
"command.completed",
]) {
expect(queryKeysForRunEvent("run-1", event, "stage-1")).toEqual([
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
}
expect(queryKeysForRunEvent("run-1", "agent.message", "stage-1")).toEqual([
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "stage-1"),
queryKeys.runs.stageContextWindow("run-1", "stage-1"),
]);
});
test("agent message without a node_id still invalidates projected state", () => {
expect(queryKeysForRunEvent("run-1", "agent.message")).toEqual([
queryKeys.runs.state("run-1"),
]);
});
});

View file

@ -111,10 +111,7 @@ export async function deleteRuns(
request?: Request,
): Promise<BatchDeleteRunsResponse> {
try {
// See `batchRunLifecycleAction` for the `as unknown as` rationale:
// openapi-generator types `uniqueItems` arrays as `Set<T>` while the wire
// contract is a JSON array.
const body = { run_ids: runIds, force } as unknown as BatchDeleteRunsRequest;
const body: BatchDeleteRunsRequest = { run_ids: runIds, force };
return await apiData(() => runsApi.batchDeleteRuns(body, requestSignalOptions(request)));
} catch (error) {
throw lifecycleActionErrorFromError(error);
@ -284,10 +281,7 @@ async function batchRunLifecycleAction(
request?: Request,
): Promise<BatchRunLifecycleResponse> {
try {
// openapi-generator's TypeScript client represents `uniqueItems` arrays as
// Set<T>, but the HTTP wire contract is still a JSON array. Keep an array
// here so Axios serializes the request body correctly.
const body = { run_ids: runIds } as unknown as BatchRunLifecycleRequest;
const body: BatchRunLifecycleRequest = { run_ids: runIds };
switch (action) {
case "archive":
return await apiData(() => runsApi.batchArchiveRuns(body, requestSignalOptions(request)));

View file

@ -85,6 +85,8 @@ describe("queryKeysForRunEvent", () => {
test("interrupt settlement invalidates projected control state and stage activity", () => {
expect(queryKeysForRunEvent("run-1", "agent.round.interrupted", "nap@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.events("run-1", 1000),
queryKeys.runs.stageEvents("run-1", "nap@1"),
@ -129,10 +131,16 @@ describe("queryKeysForRunEvent", () => {
test("every inference projection transition invalidates live run state", () => {
for (const event of [
"agent.llm.started",
"agent.llm.first_output",
"agent.llm.retry",
"agent.error",
]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
for (const event of ["agent.llm.first_output", "agent.llm.retry"]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.state("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
@ -141,15 +149,47 @@ describe("queryKeysForRunEvent", () => {
expect(
queryKeysForRunEvent("run-1", "agent.message", "code@1"),
).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
expect(queryKeysForRunEvent("run-1", "agent.session.ended")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
]);
});
test("ACP timing events invalidate live summaries and stage events", () => {
for (const event of [
"agent.acp.started",
"agent.acp.completed",
"agent.acp.cancelled",
"agent.acp.timed_out",
]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
]);
}
});
test("tool timing events invalidate live summaries and stage resources", () => {
for (const event of ["agent.tool.started", "agent.tool.completed"]) {
expect(queryKeysForRunEvent("run-1", event, "code@1")).toEqual([
queryKeys.runs.detail("run-1"),
queryKeys.runs.state("run-1"),
queryKeys.runs.billing("run-1"),
queryKeys.runs.stageEvents("run-1", "code@1"),
queryKeys.runs.stageContextWindow("run-1", "code@1"),
]);
}
});
test("watchdog timeout refreshes the stage events for that stage", () => {
expect(
queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"),

View file

@ -118,6 +118,22 @@ const INFERENCE_EVENTS = new Set([
"agent.error",
"agent.session.ended",
]);
const INFERENCE_TIMING_EVENTS = new Set([
"agent.llm.started",
"agent.message",
"agent.error",
"agent.session.ended",
]);
const TOOL_TIMING_EVENTS = new Set([
"agent.tool.started",
"agent.tool.completed",
]);
const ACP_TIMING_EVENTS = new Set([
"agent.acp.started",
"agent.acp.completed",
"agent.acp.cancelled",
"agent.acp.timed_out",
]);
// Todo / task mutation events refresh `getRunState` consumers (so per-stage
// todo projections update live) and the run events list.
const TODO_EVENTS = new Set([
@ -126,6 +142,14 @@ const TODO_EVENTS = new Set([
"todo.deleted",
]);
function liveTimingKeys(runId: string): Key[] {
return [
queryKeys.runs.detail(runId),
queryKeys.runs.state(runId),
queryKeys.runs.billing(runId),
];
}
export function queryKeysForRunEvent(
runId: string,
event: string,
@ -184,6 +208,12 @@ export function queryKeysForRunEvent(
if (AGENT_CONTROL_STATE_EVENTS.has(event)) {
keys.unshift(queryKeys.runs.state(runId));
}
if (event === "agent.round.interrupted") {
keys.unshift(
queryKeys.runs.detail(runId),
queryKeys.runs.billing(runId),
);
}
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
@ -192,7 +222,9 @@ export function queryKeysForRunEvent(
}
if (INFERENCE_EVENTS.has(event)) {
const keys: Key[] = [queryKeys.runs.state(runId)];
const keys = INFERENCE_TIMING_EVENTS.has(event)
? liveTimingKeys(runId)
: [queryKeys.runs.state(runId)];
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
if (event === "agent.message") {
@ -202,6 +234,23 @@ export function queryKeysForRunEvent(
return keys;
}
if (TOOL_TIMING_EVENTS.has(event)) {
const keys = liveTimingKeys(runId);
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
}
return keys;
}
if (ACP_TIMING_EVENTS.has(event)) {
const keys = liveTimingKeys(runId);
if (stageId) {
keys.push(queryKeys.runs.stageEvents(runId, stageId));
}
return keys;
}
if (event === "watchdog.timeout") {
return stageId ? [queryKeys.runs.stageEvents(runId, stageId)] : [];
}

View file

@ -3,6 +3,7 @@ import type { PaginatedRunStageList, StageHandler, StageState } from "@qltysh/fa
import type { Stage } from "../components/stage-sidebar";
import { aggregateGraphNodeStatus, formatStageLabel, mapRunStagesToSidebarStages } from "./stage-sidebar";
import { makeBilledTokenCounts } from "./test-fixtures";
function makeStage(nodeId: string, visit: number, status: StageState): Stage {
return {
@ -17,6 +18,7 @@ function makeStage(nodeId: string, visit: number, status: StageState): Stage {
duration: "--",
startedAt: null,
providerUsed: null,
billing: makeBilledTokenCounts(),
};
}
@ -38,6 +40,15 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
},
billing: makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
total_tokens: 43_690,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_usd_micros: 720_000,
}),
},
{
id: "apply-changes@2",
@ -46,6 +57,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "apply",
visit: 2,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -64,6 +76,10 @@ describe("mapRunStagesToSidebarStages", () => {
model: "gpt-5.5",
reasoning_effort: "high",
});
// Each visit keeps its own tokens and cost, so the stage popover never
// shows a sibling visit's usage.
expect(result[0].billing.total_usd_micros).toBe(720_000);
expect(result[1].billing.total_usd_micros).toBeUndefined();
expect(formatStageLabel(result[0])).toBe("Apply Changes");
expect(result[1].id).toBe("apply-changes@2");
@ -83,6 +99,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "start",
visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "verify@1",
@ -91,6 +108,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "exit@1",
@ -99,6 +117,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "exit",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -118,6 +137,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "running",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -137,6 +157,7 @@ describe("mapRunStagesToSidebarStages", () => {
node_id: "work",
visit: 1,
graph_visit: 1,
billing: makeBilledTokenCounts(),
},
{
id: "work@2",
@ -147,6 +168,7 @@ describe("mapRunStagesToSidebarStages", () => {
visit: 2,
graph_visit: 1,
resumed_from_stage_id: "work@1",
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -172,6 +194,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "succeeded",
node_id: "verify",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },
@ -192,6 +215,7 @@ describe("mapRunStagesToSidebarStages", () => {
status: "pending",
node_id: "approval",
visit: 1,
billing: makeBilledTokenCounts(),
},
],
meta: { has_more: false },

View file

@ -1,5 +1,6 @@
import { StageState } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
PaginatedRunStageList,
StageHandler,
StageModelUsage,
@ -27,6 +28,11 @@ export interface Stage {
resumedFromStageId: string | null;
startedAt: string | null;
providerUsed: StageModelUsage | null;
/**
* Tokens and cost for this visit alone, priced the same way the Billing tab
* prices its per-node rows. All-zero counts mean the stage called no model.
*/
billing: BilledTokenCounts;
}
export const ACTIVE_STAGE_STATES: ReadonlySet<StageState> = new Set([
@ -102,6 +108,7 @@ export function mapRunStagesToSidebarStages(
: "--",
startedAt: stage.started_at ?? null,
providerUsed: stage.provider_used ?? null,
billing: stage.billing,
});
}
return stages;

View file

@ -1,4 +1,4 @@
import type { Principal } from "@qltysh/fabro-api-client";
import type { BilledTokenCounts, Principal } from "@qltysh/fabro-api-client";
export const TEST_PRINCIPAL: Principal = {
kind: "user",
@ -6,3 +6,17 @@ export const TEST_PRINCIPAL: Principal = {
login: "test",
auth_method: "dev_token",
};
export function makeBilledTokenCounts(
overrides: Partial<BilledTokenCounts> = {},
): BilledTokenCounts {
return {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 0,
output_tokens: 0,
reasoning_tokens: 0,
total_tokens: 0,
...overrides,
};
}

View file

@ -1,14 +1,18 @@
import { useMemo } from "react";
import { useParams } from "react-router";
import { ArrowDownTrayIcon, PaperClipIcon } from "@heroicons/react/24/outline";
import { Disclosure, DisclosureButton, DisclosurePanel } from "@headlessui/react";
import { ArrowDownTrayIcon, ChevronRightIcon, PaperClipIcon } from "@heroicons/react/24/outline";
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import { EmptyState, ErrorState, LoadingState } from "../components/state";
import { StageSidebar } from "../components/stage-sidebar";
import { stageArtifactDownloadUrl } from "../lib/api-client";
import { formatBytes } from "../lib/format";
import { plural } from "../lib/plural";
import { useRunArtifacts, useRunStages } from "../lib/queries";
import { formatStageLabel, mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
import { mapRunStagesToSidebarStages } from "../lib/stage-sidebar";
import type { ArtifactFile, ArtifactVersion } from "./run-artifacts/group";
import { groupArtifactsByFile } from "./run-artifacts/group";
export const handle = { wide: true };
@ -25,7 +29,12 @@ export default function RunArtifacts() {
<div className="flex gap-6">
<StageSidebar stages={stages} runId={id!} activeLink="artifacts" />
<div className="min-w-0 flex-1">
<RunArtifactsBody runId={id!} artifactsQuery={artifactsQuery} stages={stages} />
<RunArtifactsBody
runId={id!}
artifactsQuery={artifactsQuery}
stagesQuery={stagesQuery}
stages={stages}
/>
</div>
</div>
);
@ -34,86 +43,40 @@ export default function RunArtifacts() {
function RunArtifactsBody({
runId,
artifactsQuery,
stagesQuery,
stages,
}: {
runId: string;
artifactsQuery: ReturnType<typeof useRunArtifacts>;
stagesQuery: ReturnType<typeof useRunStages>;
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
}) {
if (artifactsQuery.error) {
const error = artifactsQuery.error ?? stagesQuery.error;
if (error) {
return (
<ErrorState
title="Couldn't load artifacts"
description={errorMessage(artifactsQuery.error)}
onRetry={() => void artifactsQuery.mutate()}
description={errorMessage(error)}
onRetry={() => {
if (artifactsQuery.error) void artifactsQuery.mutate();
if (stagesQuery.error) void stagesQuery.mutate();
}}
/>
);
}
if (artifactsQuery.data === undefined) {
if (artifactsQuery.data === undefined || stagesQuery.data === undefined) {
return <LoadingState label="Loading artifacts…" />;
}
const entries = artifactsQuery.data?.data ?? [];
if (entries.length === 0) {
return (
<EmptyState
icon={PaperClipIcon}
title="No artifacts captured"
description="No stage in this run produced any artifacts."
/>
);
}
return <ArtifactList runId={runId} entries={entries} stages={stages} />;
return (
<ArtifactFiles
runId={runId}
entries={artifactsQuery.data?.data ?? []}
stages={stages}
/>
);
}
interface StageGroup {
key: string;
stageId: string;
retry: number;
label: string;
entries: RunArtifactEntry[];
totalBytes: number;
}
function groupArtifacts(
entries: readonly RunArtifactEntry[],
stages: ReturnType<typeof mapRunStagesToSidebarStages>,
): StageGroup[] {
const stageLabels = new Map<string, string>();
for (const stage of stages) {
stageLabels.set(stage.id, formatStageLabel(stage));
}
const groups = new Map<string, StageGroup>();
for (const entry of entries) {
const key = `${entry.stage_id}#${entry.retry}`;
const existing = groups.get(key);
if (existing) {
existing.entries.push(entry);
existing.totalBytes += entry.size;
} else {
groups.set(key, {
key,
stageId: entry.stage_id,
retry: entry.retry,
label: stageLabels.get(entry.stage_id) ?? entry.node_slug,
entries: [entry],
totalBytes: entry.size,
});
}
}
for (const group of groups.values()) {
group.entries.sort((a, b) => a.relative_path.localeCompare(b.relative_path));
}
const sortedGroups = Array.from(groups.values());
sortedGroups.sort((a, b) => {
const labelCmp = a.label.localeCompare(b.label);
return labelCmp !== 0 ? labelCmp : a.retry - b.retry;
});
return sortedGroups;
}
function ArtifactList({
function ArtifactFiles({
runId,
entries,
stages,
@ -122,97 +85,209 @@ function ArtifactList({
entries: readonly RunArtifactEntry[];
stages: ReturnType<typeof mapRunStagesToSidebarStages>;
}) {
const groups = useMemo(() => groupArtifacts(entries, stages), [entries, stages]);
const totalBytes = useMemo(
() => entries.reduce((sum, entry) => sum + entry.size, 0),
[entries],
);
const files = useMemo(() => groupArtifactsByFile(entries, stages), [entries, stages]);
if (files.length === 0) {
return (
<EmptyState
icon={PaperClipIcon}
title="No artifacts captured"
description="No stage in this run produced any artifacts."
/>
);
}
return <ArtifactList runId={runId} files={files} />;
}
function ArtifactList({ runId, files }: { runId: string; files: readonly ArtifactFile[] }) {
const { captures, latestBytes, storedBytes } = useMemo(() => {
let captures = 0;
let latestBytes = 0;
let storedBytes = 0;
for (const file of files) {
captures += file.versions.length;
latestBytes += file.versions[0].size;
for (const version of file.versions) storedBytes += version.size;
}
return { captures, latestBytes, storedBytes };
}, [files]);
// Only mention versions once some file actually has more than one.
const versioned = captures > files.length;
return (
<div className="space-y-4">
<div className="flex items-baseline justify-between">
<div className="flex flex-wrap items-baseline justify-between gap-4">
<h2 className="text-sm font-medium text-fg">
{entries.length} {entries.length === 1 ? "artifact" : "artifacts"}
{files.length} {plural(files.length, "file", "files")}
{versioned && (
<span className="font-normal text-fg-muted">
{" "}
· {captures} {plural(captures, "version", "versions")}
</span>
)}
</h2>
<span className="text-xs tabular-nums text-fg-muted">
{formatBytes(totalBytes)} total
<span className="text-xs text-fg-muted tabular-nums">
{versioned
? `${formatBytes(latestBytes)} latest · ${formatBytes(storedBytes)} stored`
: `${formatBytes(latestBytes)} total`}
</span>
</div>
{groups.map((group) => (
<StageGroupCard key={group.key} runId={runId} group={group} />
))}
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
{files.map((file) => (
<ArtifactFileRow key={file.path} runId={runId} file={file} />
))}
</section>
</div>
);
}
function StageGroupCard({ runId, group }: { runId: string; group: StageGroup }) {
function ArtifactFileRow({ runId, file }: { runId: string; file: ArtifactFile }) {
const hasEarlier = file.versions.length > 1;
const latest = file.versions[0];
return (
<section className="overflow-hidden rounded-md border border-line bg-panel-alt">
<header className="flex items-baseline justify-between border-b border-line px-4 py-2.5">
<div className="flex items-baseline gap-2">
<h3 className="text-sm font-medium text-fg">{group.label}</h3>
{group.retry > 0 && (
<span className="rounded bg-overlay px-1.5 py-0.5 text-[11px] font-medium text-fg-3">
retry {group.retry}
<Disclosure as="div" className="border-t border-line first:border-t-0">
{({ open }) => (
<>
<div className="flex items-center gap-2 px-3 py-2.5 sm:gap-4 sm:px-4">
{hasEarlier ? (
<DisclosureButton className="group shrink-0 rounded-md p-1 text-fg-3 transition-colors hover:bg-overlay hover:text-fg-2 focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500">
<span className="sr-only">
{open ? "Hide" : "Show"} earlier versions of {file.name}
</span>
<ChevronRightIcon
className="size-3.5 transition-transform group-data-open:rotate-90"
aria-hidden="true"
/>
</DisclosureButton>
) : (
<span className="size-5 shrink-0" aria-hidden="true" />
)}
<span className="min-w-0 flex-1" title={file.path}>
<span className="block truncate font-mono text-xs">
<span className="text-fg-muted">{file.dir}</span>
<span className="text-fg-2">{file.name}</span>
</span>
<span className="mt-0.5 block truncate text-[11px] text-fg-3 md:hidden">
<VersionLabel version={latest} />
</span>
</span>
{hasEarlier && (
<span className="hidden shrink-0 rounded-full bg-overlay-strong px-2 py-0.5 text-[11px] text-fg-3 lg:inline">
{file.versions.length}{" "}
{plural(file.versions.length, "version", "versions")}
</span>
)}
<span className="hidden max-w-48 shrink-0 truncate text-xs text-fg-3 md:inline">
<VersionLabel version={latest} />
</span>
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
{formatBytes(latest.size)}
</span>
<DownloadLink runId={runId} file={file} version={latest} />
</div>
{hasEarlier && (
<DisclosurePanel
as="ul"
className="border-t border-line bg-black/15 py-1"
>
<EarlierVersions runId={runId} file={file} />
</DisclosurePanel>
)}
</div>
<span className="text-xs tabular-nums text-fg-muted">
{group.entries.length} {group.entries.length === 1 ? "file" : "files"}
{" · "}
{formatBytes(group.totalBytes)}
</span>
</header>
<ul className="divide-y divide-line">
{group.entries.map((entry) => (
<ArtifactRow
key={`${group.key}#${entry.relative_path}`}
runId={runId}
entry={entry}
/>
))}
</ul>
</section>
</>
)}
</Disclosure>
);
}
function ArtifactRow({ runId, entry }: { runId: string; entry: RunArtifactEntry }) {
function EarlierVersions({ runId, file }: { runId: string; file: ArtifactFile }) {
return (
<>
{file.versions.map((version, index) =>
index === 0 ? null : (
<li
key={`${version.stageId}#${version.retry}`}
className="flex items-center gap-2 py-1.5 pr-3 pl-10 hover:bg-overlay sm:gap-4 sm:pr-4 sm:pl-14"
>
<span className="min-w-0 flex-1 truncate text-xs text-fg-3">
<VersionLabel version={version} />
</span>
<span className="shrink-0 text-xs text-fg-muted tabular-nums">
{formatBytes(version.size)}
</span>
<SizeDelta delta={version.delta} />
<DownloadLink runId={runId} file={file} version={version} />
</li>
),
)}
</>
);
}
function VersionLabel({ version }: { version: ArtifactVersion }) {
const attempt = attemptLabel(version);
return (
<>
{version.stageLabel}
{attempt && <span className="ml-2 text-fg-muted">{attempt}</span>}
</>
);
}
function attemptLabel(version: ArtifactVersion): string | null {
return version.retry > 1 ? `attempt ${version.retry}` : null;
}
function SizeDelta({ delta }: { delta: number | null }) {
if (delta === null) {
return <span className="shrink-0 text-[11px] text-fg-muted tabular-nums">first</span>;
}
const tone = delta < 0 ? "text-amber" : "text-mint";
const sign = delta < 0 ? "−" : "+";
return (
<span className={`shrink-0 text-[11px] ${tone} tabular-nums`}>
{sign}
{formatBytes(Math.abs(delta))}
</span>
);
}
function DownloadLink({
runId,
file,
version,
}: {
runId: string;
file: ArtifactFile;
version: ArtifactVersion;
}) {
const href = stageArtifactDownloadUrl(
runId,
entry.stage_id,
entry.relative_path,
entry.retry,
version.stageId,
file.path,
version.retry,
);
const attempt = attemptLabel(version);
const source = attempt ? `${version.stageLabel}, ${attempt}` : version.stageLabel;
return (
<li className="flex items-center gap-4 px-4 py-2">
<span
className="flex-1 truncate font-mono text-xs text-fg-2"
title={entry.relative_path}
>
{entry.relative_path}
</span>
<span className="shrink-0 tabular-nums text-xs text-fg-muted">
{formatBytes(entry.size)}
</span>
<a
href={href}
download={basename(entry.relative_path)}
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
>
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
Download
</a>
</li>
<a
href={href}
download={file.name}
aria-label={`Download ${file.name} from ${source}`}
className="inline-flex shrink-0 items-center gap-1 rounded-md px-2 py-1 text-xs text-fg-3 transition-colors hover:bg-overlay hover:text-fg focus-visible:outline-2 focus-visible:-outline-offset-1 focus-visible:outline-teal-500"
>
<ArrowDownTrayIcon className="size-3.5" aria-hidden="true" />
<span className="hidden sm:inline">Download</span>
</a>
);
}
function basename(path: string): string {
const idx = path.lastIndexOf("/");
return idx >= 0 ? path.slice(idx + 1) : path;
}
function errorMessage(error: unknown): string | undefined {
return error instanceof Error ? error.message : undefined;
}

View file

@ -0,0 +1,192 @@
import { describe, expect, test } from "bun:test";
import { StageHandler, StageState } from "@qltysh/fabro-api-client";
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import type { Stage } from "../../lib/stage-sidebar";
import { groupArtifactsByFile, splitArtifactPath } from "./group";
function stage(nodeId: string, startedAt: string | null, visit = 1): Stage {
return {
id: `${nodeId}@${visit}`,
name: nodeId,
handler: StageHandler.AGENT,
nodeId,
visit,
graphVisit: null,
resumedFromStageId: null,
status: StageState.SUCCEEDED,
duration: "1s",
startedAt,
providerUsed: null,
};
}
function artifact(
nodeSlug: string,
path: string,
size: number,
retry = 1,
visit = 1,
): RunArtifactEntry {
return {
stage_id: `${nodeSlug}@${visit}`,
node_slug: nodeSlug,
retry,
relative_path: path,
size,
};
}
/** Mirrors run 01KYJ8ZR0N: one report rewritten by four stages. */
const REPORT = ".ai/reports/2026-07-27-wrk-002-instance-lifecycle.md";
const STAGES: Stage[] = [
stage("start", "2026-07-27T17:12:08Z"),
stage("plan", "2026-07-27T17:21:44Z"),
stage("implement_plan", "2026-07-27T17:44:18Z"),
stage("simplify", "2026-07-27T18:43:11Z"),
stage("consolidate_reviews", "2026-07-27T19:44:26Z"),
stage("fix_review_findings", "2026-07-27T19:49:22Z"),
];
describe("splitArtifactPath", () => {
test("splits a nested path into directory prefix and filename", () => {
expect(splitArtifactPath(".ai/reports/run.md")).toEqual({
dir: ".ai/reports/",
name: "run.md",
});
});
test("leaves a root-level path without a directory", () => {
expect(splitArtifactPath("README.md")).toEqual({ dir: "", name: "README.md" });
});
});
describe("groupArtifactsByFile", () => {
test("collapses repeated captures of one path into a single file", () => {
const files = groupArtifactsByFile(
[
artifact("consolidate_reviews", REPORT, 14323),
artifact("fix_review_findings", REPORT, 17483),
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
],
STAGES,
);
expect(files).toHaveLength(1);
expect(files[0].path).toBe(REPORT);
expect(files[0].dir).toBe(".ai/reports/");
expect(files[0].name).toBe("2026-07-27-wrk-002-instance-lifecycle.md");
expect(files[0].versions).toHaveLength(4);
});
test("orders versions newest first using the API stage order", () => {
const stages = [
stage("implement_plan", "2026-07-27T20:00:00Z"),
stage("simplify", "2026-07-27T18:43:11Z"),
];
const files = groupArtifactsByFile(
[
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
],
stages,
);
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
"simplify",
"implement_plan",
]);
expect(files[0].versions[0].size).toBe(13162);
});
test.each([
["equal", "2026-07-27T18:43:11Z", "2026-07-27T18:43:11Z"],
["missing", null, null],
])("preserves API order when stage timestamps are %s", (_case, firstAt, secondAt) => {
const stages = [
stage("implement_plan", firstAt),
stage("simplify", secondAt),
];
const files = groupArtifactsByFile(
[
artifact("simplify", REPORT, 13162),
artifact("implement_plan", REPORT, 8422),
],
stages,
);
expect(files[0].versions.map((v) => v.stageLabel)).toEqual([
"simplify",
"implement_plan",
]);
});
test("reports the byte change each capture introduced, oldest capture first", () => {
const files = groupArtifactsByFile(
[
artifact("implement_plan", REPORT, 8422),
artifact("simplify", REPORT, 13162),
artifact("consolidate_reviews", REPORT, 14323),
artifact("fix_review_findings", REPORT, 17483),
],
STAGES,
);
// versions are newest-first, so deltas read 17483-14323, 14323-13162, ...
expect(files[0].versions.map((v) => v.delta)).toEqual([3160, 1161, 4740, null]);
});
test("drops captures from graph control nodes", () => {
const files = groupArtifactsByFile(
[
artifact("start", ".ai/reports/pre-existing.md", 12402),
artifact("plan", ".ai/plans/plan.md", 21749),
],
STAGES,
);
expect(files.map((file) => file.path)).toEqual([".ai/plans/plan.md"]);
});
test("sorts files by their most recent capture", () => {
const files = groupArtifactsByFile(
[
artifact("plan", ".ai/plans/plan.md", 21749),
artifact("fix_review_findings", REPORT, 17483),
artifact("simplify", ".ai/reviews/bugs.xml", 5231),
],
STAGES,
);
expect(files.map((file) => file.path)).toEqual([
REPORT,
".ai/reviews/bugs.xml",
".ai/plans/plan.md",
]);
});
test("keeps retries of one stage as separate ordered versions", () => {
const files = groupArtifactsByFile(
[
artifact("simplify", REPORT, 13162, 2),
artifact("simplify", REPORT, 9000, 1),
],
STAGES,
);
expect(files[0].versions.map((v) => v.retry)).toEqual([2, 1]);
expect(files[0].versions[0].size).toBe(13162);
expect(files[0].versions.map((v) => v.delta)).toEqual([4162, null]);
});
test("returns no files when every capture came from a control node", () => {
const files = groupArtifactsByFile(
[artifact("start", ".ai/reports/pre-existing.md", 12402)],
STAGES,
);
expect(files).toEqual([]);
});
});

View file

@ -0,0 +1,104 @@
import type { RunArtifactEntry } from "@qltysh/fabro-api-client";
import { isVisibleStage } from "../../data/runs";
import type { Stage } from "../../lib/stage-sidebar";
import { formatStageLabel } from "../../lib/stage-sidebar";
/** One capture of a file, written by a single stage attempt. */
export interface ArtifactVersion {
stageId: string;
stageLabel: string;
retry: number;
size: number;
/** Byte change this capture introduced; null for the first capture. */
delta: number | null;
}
/** One artifact path together with its capture history, newest first. */
export interface ArtifactFile {
path: string;
/** Directory prefix including the trailing slash, or "" at the root. */
dir: string;
name: string;
versions: readonly [ArtifactVersion, ...ArtifactVersion[]];
}
export function splitArtifactPath(path: string): { dir: string; name: string } {
const idx = path.lastIndexOf("/");
return idx >= 0
? { dir: path.slice(0, idx + 1), name: path.slice(idx + 1) }
: { dir: "", name: path };
}
interface StageInfo {
label: string;
order: number;
}
/** Stage display data keyed by ID, preserving the API's event order. */
function stageInfoById(stages: readonly Stage[]): Map<string, StageInfo> {
const info = new Map<string, StageInfo>();
stages.forEach((stage, order) => {
info.set(stage.id, { label: formatStageLabel(stage), order });
});
return info;
}
/**
* Collapse raw `(stage, retry, path)` capture keys into one entry per file,
* carrying the ordered history of every capture of that path.
*
* Captures from graph control nodes (`start`, `exit`) are dropped: those nodes
* run no work, so anything they match is a pre-existing workspace file rather
* than something the run produced.
*/
export function groupArtifactsByFile(
entries: readonly RunArtifactEntry[],
stages: readonly Stage[],
): ArtifactFile[] {
const stageInfo = stageInfoById(stages);
const byPath = new Map<string, [ArtifactVersion, ...ArtifactVersion[]]>();
for (const entry of entries) {
if (!isVisibleStage(entry.node_slug)) continue;
const info = stageInfo.get(entry.stage_id);
const version: ArtifactVersion = {
stageId: entry.stage_id,
stageLabel: info?.label ?? entry.node_slug,
retry: entry.retry,
size: entry.size,
delta: null,
};
const bucket = byPath.get(entry.relative_path);
if (bucket) bucket.push(version);
else byPath.set(entry.relative_path, [version]);
}
const files: Array<{ file: ArtifactFile; order: number }> = [];
for (const [path, versions] of byPath) {
// Oldest first, so each version's delta is the change that capture introduced.
versions.sort(
(a, b) =>
(stageInfo.get(a.stageId)?.order ?? -1) -
(stageInfo.get(b.stageId)?.order ?? -1) ||
a.retry - b.retry ||
a.stageId.localeCompare(b.stageId),
);
versions.forEach((version, index) => {
version.delta = index === 0 ? null : version.size - versions[index - 1].size;
});
versions.reverse();
const latest = versions[0];
const { dir, name } = splitArtifactPath(path);
files.push({
file: { path, dir, name, versions },
order: stageInfo.get(latest.stageId)?.order ?? -1,
});
}
// Most recently written file first — the page answers "what just happened?".
files.sort((a, b) => b.order - a.order || a.file.path.localeCompare(b.file.path));
return files.map((entry) => entry.file);
}

View file

@ -2,11 +2,12 @@ import { afterEach, describe, expect, mock, test } from "bun:test";
import TestRenderer from "react-test-renderer";
import type {
BilledTokenCounts,
RunBilling,
StageTiming,
} from "@qltysh/fabro-api-client";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
function stageTiming(wall_time_ms = 0, inference_time_ms = 0, tool_time_ms = 0): StageTiming {
return {
wall_time_ms,
@ -24,25 +25,12 @@ mock.module("../lib/queries", () => ({
const { default: RunBillingRoute } = await import("./run-billing");
function zeroBilling(overrides: Partial<BilledTokenCounts> = {}): BilledTokenCounts {
return {
cache_read_tokens: 0,
cache_write_tokens: 0,
input_tokens: 0,
output_tokens: 0,
reasoning_tokens: 0,
total_tokens: 0,
total_usd_micros: null,
...overrides,
};
}
function billing(overrides: Partial<RunBilling> = {}): RunBilling {
return {
stages: [],
totals: {
timing: stageTiming(),
...zeroBilling(),
...makeBilledTokenCounts(),
},
by_model: [],
...overrides,
@ -86,21 +74,21 @@ describe("RunBilling", () => {
{
stage: { id: "start", name: "start" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(),
state: "succeeded",
},
{
stage: { id: "command", name: "command" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(61000),
state: "succeeded",
},
],
totals: {
timing: stageTiming(61000),
...zeroBilling(),
...makeBilledTokenCounts(),
},
}),
);
@ -121,7 +109,7 @@ describe("RunBilling", () => {
{
stage: { id: "start", name: "start" },
model: null,
billing: zeroBilling(),
billing: makeBilledTokenCounts(),
timing: stageTiming(),
state: "succeeded",
},
@ -131,7 +119,7 @@ describe("RunBilling", () => {
provider: "anthropic",
model_id: "claude-sonnet-4-5",
},
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -143,7 +131,7 @@ describe("RunBilling", () => {
],
totals: {
timing: stageTiming(42000),
...zeroBilling({
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -157,7 +145,7 @@ describe("RunBilling", () => {
model_id: "claude-sonnet-4-5",
},
stages: 1,
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -204,7 +192,7 @@ describe("RunBilling", () => {
model_id: "claude-opus-4-6",
speed: "fast",
},
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -217,7 +205,7 @@ describe("RunBilling", () => {
],
totals: {
timing: stageTiming(),
...zeroBilling({
...makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -232,7 +220,7 @@ describe("RunBilling", () => {
speed: "fast",
},
stages: 1,
billing: zeroBilling({
billing: makeBilledTokenCounts({
input_tokens: 1200,
output_tokens: 300,
total_tokens: 1500,
@ -268,4 +256,4 @@ describe("RunBilling", () => {
Date.now = originalNow;
}
});
});
});

View file

@ -2,6 +2,11 @@ import { Fragment, useMemo } from "react";
import { EmptyState } from "../components/state";
import { Tooltip } from "../components/ui";
import {
billableOutputTokens,
billingTokenBuckets,
hasBillingUsage,
} from "../lib/billing";
import {
formatDurationMs,
formatTokenCount,
@ -11,6 +16,7 @@ import { useRunBilling } from "../lib/queries";
import { IN_FLIGHT_STAGE_STATES } from "../lib/stage-sidebar";
import { useTickingNow } from "../lib/time";
import type {
BilledTokenCounts,
BillingModelRef,
RunBilling,
RunBillingStage,
@ -39,23 +45,15 @@ function isInFlight(stage: RunBillingStage): boolean {
function isVisibleRow(row: MappedStageRow): boolean {
if (row.inFlight) return true;
return (
(row.inputTokens ?? 0) > 0 ||
(row.outputTokens ?? 0) > 0 ||
(row.totalUsdMicros ?? 0) > 0
);
return row.billing != null && hasBillingUsage(row.billing);
}
interface MappedStageRow {
stage: string;
model: string | null;
inputTokens: number | null;
outputTokens: number | null;
cacheReadTokens: number | null;
cacheWriteTokens: number | null;
wallTimeMs: number;
totalUsdMicros: number | null | undefined;
inFlight: boolean;
stage: string;
model: string | null;
billing: BilledTokenCounts | null;
wallTimeMs: number;
inFlight: boolean;
}
function liveWallTimeMs(stage: RunBillingStage, now: number): number {
@ -73,49 +71,28 @@ export const handle = { wide: true };
function mapStageRow(stage: RunBillingStage, wallTimeMs: number): MappedStageRow {
const hasModel = stage.model != null;
return {
stage: stage.stage.name,
model: formatModelRef(stage.model),
inputTokens: hasModel ? stage.billing.input_tokens : null,
outputTokens: hasModel
? stage.billing.output_tokens + stage.billing.reasoning_tokens
: null,
cacheReadTokens: hasModel ? stage.billing.cache_read_tokens : null,
cacheWriteTokens: hasModel ? stage.billing.cache_write_tokens : null,
stage: stage.stage.name,
model: formatModelRef(stage.model),
billing: hasModel ? stage.billing : null,
wallTimeMs,
totalUsdMicros: stage.billing.total_usd_micros,
inFlight: isInFlight(stage),
inFlight: isInFlight(stage),
};
}
/** Hover breakdown of the disjoint token buckets behind an `in / out` count. */
function TokenBreakdown({
cacheReadTokens,
cacheWriteTokens,
inputTokens,
outputTokens,
}: {
cacheReadTokens: number;
cacheWriteTokens: number;
inputTokens: number;
outputTokens: number;
}) {
const rows = [
{ label: "Cache read", value: cacheReadTokens },
{ label: "Cache creation", value: cacheWriteTokens },
{ label: "Uncached", value: inputTokens },
{ label: "Output", value: outputTokens },
];
function TokenBreakdown({ billing }: { billing: BilledTokenCounts }) {
const buckets = billingTokenBuckets(billing);
return (
<div className="min-w-44 py-0.5">
<div className="mb-1.5 border-b border-line pb-1 font-medium text-fg-2">
<div className="border-line text-fg-2 mb-1.5 border-b pb-1 font-medium">
Tokens in / out
</div>
<dl className="grid grid-cols-[1fr_auto] gap-x-6 gap-y-1">
{rows.map((row) => (
<Fragment key={row.label}>
<dt className="text-fg-3">{row.label}</dt>
<dd className="text-right font-mono tabular-nums text-fg">
{formatTokens(row.value)}
{buckets.map((bucket) => (
<Fragment key={bucket.label}>
<dt className="text-fg-3">{bucket.label}</dt>
<dd className="text-fg text-right font-mono tabular-nums">
{formatTokens(bucket.value)}
</dd>
</Fragment>
))}
@ -128,42 +105,16 @@ function TokenBreakdown({
* Renders an `input / output` token count. When the row has model usage,
* hovering the count reveals the cache breakdown.
*/
function TokensCell({
inputTokens,
outputTokens,
cacheReadTokens,
cacheWriteTokens,
}: {
inputTokens: number | null;
outputTokens: number | null;
cacheReadTokens: number | null;
cacheWriteTokens: number | null;
}) {
function TokensCell({ billing }: { billing: BilledTokenCounts | null }) {
const display = (
<>
{formatTokens(inputTokens)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(outputTokens)}
{formatTokens(billing?.input_tokens)} <span className="text-fg-muted">/</span>{" "}
{formatTokens(billing ? billableOutputTokens(billing) : null)}
</>
);
if (
inputTokens == null ||
outputTokens == null ||
cacheReadTokens == null ||
cacheWriteTokens == null
) {
return display;
}
if (!billing) return display;
return (
<Tooltip
label={
<TokenBreakdown
cacheReadTokens={cacheReadTokens}
cacheWriteTokens={cacheWriteTokens}
inputTokens={inputTokens}
outputTokens={outputTokens}
/>
}
>
<Tooltip label={<TokenBreakdown billing={billing} />}>
<span>{display}</span>
</Tooltip>
);
@ -189,15 +140,14 @@ export default function RunBilling({ params }: { params: { id: string } }) {
if (!billing) return [];
return billing.by_model
.map((entry) => ({
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
inputTokens: entry.billing.input_tokens,
outputTokens: entry.billing.output_tokens + entry.billing.reasoning_tokens,
cacheReadTokens: entry.billing.cache_read_tokens,
cacheWriteTokens: entry.billing.cache_write_tokens,
totalUsdMicros: entry.billing.total_usd_micros,
model: formatModelRef(entry.model) ?? EMPTY_VALUE,
stages: entry.stages,
billing: entry.billing,
}))
.sort((a, b) => (b.totalUsdMicros ?? -1) - (a.totalUsdMicros ?? -1));
.sort(
(a, b) =>
(b.billing.total_usd_micros ?? -1) - (a.billing.total_usd_micros ?? -1),
);
}, [billing]);
// Re-derive only the in-flight rows on each tick; everything else stays put.
@ -218,16 +168,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
: (billing?.totals.timing.wall_time_ms ?? 0);
const hasLlmStages = (billing?.by_model.length ?? 0) > 0;
const totalInput = hasLlmStages ? (billing?.totals.input_tokens ?? null) : null;
const totalOutput = hasLlmStages && billing
? billing.totals.output_tokens + billing.totals.reasoning_tokens
: null;
const totalCacheRead = hasLlmStages
? (billing?.totals.cache_read_tokens ?? null)
: null;
const totalCacheWrite = hasLlmStages
? (billing?.totals.cache_write_tokens ?? null)
: null;
const totalBilling = hasLlmStages && billing ? billing.totals : null;
const totalUsdMicros = billing?.totals.total_usd_micros;
const modelStageCount = modelBreakdown.reduce((sum, row) => sum + row.stages, 0);
const visibleRows = rows.filter(isVisibleRow);
@ -268,18 +209,13 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.model ?? EMPTY_VALUE}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell
inputTokens={row.inputTokens}
outputTokens={row.outputTokens}
cacheReadTokens={row.cacheReadTokens}
cacheWriteTokens={row.cacheWriteTokens}
/>
<TokensCell billing={row.billing} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatDurationMs(row.wallTimeMs)}
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.totalUsdMicros)}
{formatUsdMicrosOrDash(row.billing?.total_usd_micros)}
</td>
</tr>
))}
@ -289,12 +225,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
<td className="px-4 py-3 font-medium text-fg">Total</td>
<td className="px-4 py-3 text-xs text-fg-muted">All models</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell
inputTokens={totalInput}
outputTokens={totalOutput}
cacheReadTokens={totalCacheRead}
cacheWriteTokens={totalCacheWrite}
/>
<TokensCell billing={totalBilling} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatDurationMs(totalWallTimeMs)}
@ -328,15 +259,10 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{row.stages}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums text-fg-3">
<TokensCell
inputTokens={row.inputTokens}
outputTokens={row.outputTokens}
cacheReadTokens={row.cacheReadTokens}
cacheWriteTokens={row.cacheWriteTokens}
/>
<TokensCell billing={row.billing} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs text-fg-3">
{formatUsdMicrosOrDash(row.totalUsdMicros)}
{formatUsdMicrosOrDash(row.billing.total_usd_micros)}
</td>
</tr>
))}
@ -348,12 +274,7 @@ export default function RunBilling({ params }: { params: { id: string } }) {
{modelStageCount}
</td>
<td className="px-4 py-3 text-right font-mono text-xs tabular-nums font-medium text-fg">
<TokensCell
inputTokens={totalInput}
outputTokens={totalOutput}
cacheReadTokens={totalCacheRead}
cacheWriteTokens={totalCacheWrite}
/>
<TokensCell billing={totalBilling} />
</td>
<td className="px-4 py-3 text-right font-mono text-xs font-medium text-fg">
{formatUsdMicrosOrDash(totalUsdMicros)}

View file

@ -183,13 +183,33 @@ import {
lifecycleActionVisibility,
} from "./run-detail/lifecycle-toasts";
const { default: RunDetail } = await import("./run-detail");
const {
default: RunDetail,
resolveDockClearance,
} = await import("./run-detail");
mock.restore();
type LifecycleToastState = import("./run-detail/lifecycle-toasts").LifecycleToastState;
type RunDetailActionResult = import("./run-detail/lifecycle-toasts").RunDetailActionResult;
const h = createElement;
describe("resolveDockClearance", () => {
test("uses only a current dock measurement", () => {
const measurement = { identity: "run_1:steer", height: 108 };
expect(resolveDockClearance(null, measurement, false)).toBe("0px");
expect(
resolveDockClearance("run_1:interview", measurement, true),
).toBe("18rem");
expect(resolveDockClearance("run_2:steer", measurement, false)).toBe(
"5rem",
);
expect(resolveDockClearance("run_1:steer", measurement, false)).toBe(
"108px",
);
});
});
function makeRunSummary({
status = "succeeded",
diffSummary = null as any,
@ -841,7 +861,7 @@ describe("RunDetail full-height child routes", () => {
});
const statuses = renderer.root.findAll(
(node) => node.type === "p" && node.props.role === "status",
(node) => node.props.role === "status",
);
expect(statuses.map(textFromTestNode)).toContain(
"Interrupted — waiting for steering",

View file

@ -19,6 +19,7 @@ import {
} from "../components/ui";
import { mutateRunListCaches } from "../lib/board-cache";
import { useDemoMode } from "../lib/demo-mode";
import { useTickingNow } from "../lib/time";
import { useSWRConfig } from "swr";
import {
useArchiveRun,
@ -56,10 +57,7 @@ import {
lifecycleActionVisibility,
updateLifecycleToastState,
} from "./run-detail/lifecycle-toasts";
import {
buildRunDetailRun,
useTickingNow,
} from "./run-detail/model";
import { buildRunDetailRun } from "./run-detail/model";
import {
buildRunDetailTabs,
childRouteLayoutFlags,
@ -69,8 +67,27 @@ import {
export const handle = { hideHeader: true };
const RUN_TIMING_REFRESH_INTERVAL_MS = 30_000;
type LifecycleTrigger = () => Promise<LifecycleMutationResult | undefined>;
export interface DockMeasurement {
identity: string;
height: number;
}
export function resolveDockClearance(
dockIdentity: string | null,
measurement: DockMeasurement | null,
hasPendingQuestions: boolean,
): string {
if (dockIdentity === null) return "0px";
if (measurement?.identity === dockIdentity) {
return `${measurement.height}px`;
}
return hasPendingQuestions ? "18rem" : "5rem";
}
export function meta({ data }: any) {
const run = data?.run;
return [{ title: run ? `${run.title} — Fabro` : "Run — Fabro" }];
@ -78,7 +95,7 @@ export function meta({ data }: any) {
export default function RunDetail({ params }: { params: { id: string } }) {
const demoMode = useDemoMode();
const runQuery = useRun(params.id);
const runQuery = useRun(params.id, RUN_TIMING_REFRESH_INTERVAL_MS);
const runStateQuery = useRunState(params.id);
const summary = runQuery.data;
const run = summary ? buildRunDetailRun(summary) : null;
@ -102,6 +119,8 @@ export default function RunDetail({ params }: { params: { id: string } }) {
const { mutate } = useSWRConfig();
const [deleteDialogOpen, setDeleteDialogOpen] = useState(false);
const [deletePending, setDeletePending] = useState(false);
const [dockMeasurement, setDockMeasurement] =
useState<DockMeasurement | null>(null);
const { push, dismiss } = useToast();
const lifecycleToastStateRef = useRef(createLifecycleToastState());
const filesCount = runQuery.data?.diff?.files_changed ?? null;
@ -116,7 +135,10 @@ export default function RunDetail({ params }: { params: { id: string } }) {
childrenCount,
});
const steerBarRef = useRef<SteerBarHandle | null>(null);
const now = useTickingNow(30_000);
const now = useTickingNow(
summary != null && summary.timestamps.completed_at == null,
RUN_TIMING_REFRESH_INTERVAL_MS,
);
const { fullHeight, hideSteerBar } = childRouteLayoutFlags(matches);
useRunEvents(params.id);
@ -144,6 +166,15 @@ export default function RunDetail({ params }: { params: { id: string } }) {
},
[handleLifecycleMutationResult],
);
const handleDockHeightChange = useCallback(
(identity: string, height: number | null) => {
setDockMeasurement((current) => {
if (height !== null) return { identity, height };
return current?.identity === identity ? null : current;
});
},
[],
);
if (runQuery.isLoading && !run) {
return <div className="py-12" />;
@ -305,7 +336,19 @@ export default function RunDetail({ params }: { params: { id: string } }) {
: []),
],
};
const dockClearance = hasPendingQuestions ? "18rem" : "5rem";
// Reserve exactly the dock's rendered height. The constants are only the
// first frame, before the dock has been measured; a fixed reservation lets
// a tall question panel cover the content it is asking about.
const dockIdentity = hasPendingQuestions
? `${params.id}:interview`
: hideSteerBar
? null
: `${params.id}:steer`;
const dockClearance = resolveDockClearance(
dockIdentity,
dockMeasurement,
hasPendingQuestions,
);
const rootStyle = {
"--fabro-interview-dock-clearance": dockClearance,
} as CSSProperties;
@ -364,14 +407,16 @@ export default function RunDetail({ params }: { params: { id: string } }) {
/>
<RunDetailDockedControls
key={dockIdentity ?? "hidden"}
runId={params.id}
hideSteerBar={hideSteerBar}
dockIdentity={dockIdentity}
hasPendingQuestions={hasPendingQuestions}
pendingQuestions={pendingQuestions}
sidebarWidth={sidebarWidth}
isResizing={isResizing}
steerBarRef={steerBarRef}
waitingForSteer={waitingForSteer}
onHeightChange={handleDockHeightChange}
/>
</div>
)}

View file

@ -0,0 +1,158 @@
import {
afterEach,
beforeEach,
describe,
expect,
mock,
test,
} from "bun:test";
import { createElement, createRef } from "react";
import TestRenderer, { act } from "react-test-renderer";
import { setupReactTestEnv } from "../../lib/test-utils";
mock.module("../../components/interview-dock", () => ({
InterviewDock: () => createElement("div", null, "Interview"),
}));
mock.module("../../components/steer-bar", () => ({
SteerBar: () => createElement("div", null, "Steer"),
}));
const { RunDetailDockedControls } = await import("./docked-controls");
mock.restore();
const mountedRenderers: TestRenderer.ReactTestRenderer[] = [];
const observerCallbacks: ResizeObserverCallback[] = [];
const dockNode = { offsetHeight: 999 };
let originalResizeObserver: typeof ResizeObserver | undefined;
let teardownReactEnv: (() => void) | undefined;
function resizeEntry(blockSize: number): ResizeObserverEntry {
return {
borderBoxSize: [{ blockSize, inlineSize: 600 }],
} as unknown as ResizeObserverEntry;
}
function renderDock({
identity,
hasPendingQuestions = false,
onHeightChange,
}: {
identity: string | null;
hasPendingQuestions?: boolean;
onHeightChange: (identity: string, height: number | null) => void;
}) {
return (
<RunDetailDockedControls
key={identity ?? "hidden"}
runId="run-1"
dockIdentity={identity}
hasPendingQuestions={hasPendingQuestions}
pendingQuestions={[]}
sidebarWidth={0}
isResizing={false}
steerBarRef={createRef()}
waitingForSteer={false}
onHeightChange={onHeightChange}
/>
);
}
beforeEach(() => {
teardownReactEnv = setupReactTestEnv();
observerCallbacks.length = 0;
originalResizeObserver = globalThis.ResizeObserver;
globalThis.ResizeObserver = class ResizeObserver {
constructor(callback: ResizeObserverCallback) {
observerCallbacks.push(callback);
}
observe() {}
unobserve() {}
disconnect() {}
} as typeof ResizeObserver;
});
afterEach(() => {
for (const renderer of mountedRenderers.splice(0)) {
act(() => renderer.unmount());
}
if (originalResizeObserver) {
globalThis.ResizeObserver = originalResizeObserver;
} else {
delete (globalThis as { ResizeObserver?: typeof ResizeObserver })
.ResizeObserver;
}
teardownReactEnv?.();
teardownReactEnv = undefined;
});
describe("RunDetailDockedControls", () => {
test("reports border-box height only when the height changes", () => {
const onHeightChange = mock(
(_identity: string, _height: number | null) => undefined,
);
let renderer!: TestRenderer.ReactTestRenderer;
act(() => {
renderer = TestRenderer.create(
renderDock({ identity: "run-1:steer", onHeightChange }),
{
createNodeMock: () => dockNode,
},
);
});
mountedRenderers.push(renderer);
const callback = observerCallbacks[0]!;
act(() => callback([resizeEntry(108)], {} as ResizeObserver));
act(() => callback([resizeEntry(108)], {} as ResizeObserver));
act(() => callback([resizeEntry(120)], {} as ResizeObserver));
expect(onHeightChange).toHaveBeenCalledTimes(2);
expect(onHeightChange.mock.calls).toEqual([
["run-1:steer", 108],
["run-1:steer", 120],
]);
});
test("clears the old identity when the dock changes or hides", () => {
const onHeightChange = mock(
(_identity: string, _height: number | null) => undefined,
);
let renderer!: TestRenderer.ReactTestRenderer;
act(() => {
renderer = TestRenderer.create(
renderDock({ identity: "run-1:steer", onHeightChange }),
{
createNodeMock: () => dockNode,
},
);
});
mountedRenderers.push(renderer);
act(() =>
observerCallbacks[0]!([resizeEntry(108)], {} as ResizeObserver),
);
act(() => {
renderer.update(
renderDock({
identity: "run-1:interview",
hasPendingQuestions: true,
onHeightChange,
}),
);
});
act(() =>
observerCallbacks[1]!([resizeEntry(220)], {} as ResizeObserver),
);
act(() => {
renderer.update(renderDock({ identity: null, onHeightChange }));
});
expect(onHeightChange.mock.calls).toEqual([
["run-1:steer", 108],
["run-1:steer", null],
["run-1:interview", 220],
["run-1:interview", null],
]);
});
});

View file

@ -1,10 +1,14 @@
import {
useCallback,
useRef,
useState,
type ReactNode,
type RefObject,
} from "react";
import { SparklesIcon } from "@heroicons/react/20/solid";
import { useResizeObserver } from "../../hooks/effects";
import AskFabroSidebar, {
SIDEBAR_WIDTH,
} from "../../components/chats/ask-fabro-sidebar";
@ -14,12 +18,12 @@ import {
SECONDARY_BUTTON_CLASS,
Tooltip,
} from "../../components/ui";
import { classNames } from "../../lib/class-names";
import {
AskFabroUnavailableReasonEnum,
type ApiQuestion,
type AskFabro,
} from "@qltysh/fabro-api-client";
import { classNames } from "./model";
const ASK_FABRO_UNAVAILABLE_TOOLTIPS: Record<
AskFabroUnavailableReasonEnum,
@ -127,32 +131,73 @@ function AskFabroTriggerButton({
export function RunDetailDockedControls({
runId,
hideSteerBar,
dockIdentity,
hasPendingQuestions,
pendingQuestions,
sidebarWidth,
isResizing,
steerBarRef,
waitingForSteer,
onHeightChange,
}: {
runId: string;
hideSteerBar: boolean;
dockIdentity: string | null;
hasPendingQuestions: boolean;
pendingQuestions: ApiQuestion[];
sidebarWidth: number;
isResizing: boolean;
steerBarRef: RefObject<SteerBarHandle | null>;
waitingForSteer: boolean;
/**
* Reports the dock's rendered height so the page can reserve exactly that
* much room beneath the scrolling content.
*/
onHeightChange: (identity: string, height: number | null) => void;
}) {
if (hideSteerBar && !hasPendingQuestions) return null;
const dockRef = useRef<HTMLDivElement | null>(null);
const reportedHeightRef = useRef<number | null>(null);
const visible = dockIdentity !== null;
const reportHeight = useCallback(
(height: number | null) => {
if (dockIdentity === null || reportedHeightRef.current === height) return;
reportedHeightRef.current = height;
onHeightChange(dockIdentity, height);
},
[dockIdentity, onHeightChange],
);
const setDockRef = useCallback(
(node: HTMLDivElement | null) => {
dockRef.current = node;
if (node === null) reportHeight(null);
},
[reportHeight],
);
useResizeObserver(
dockRef,
(entries) => {
const entry = entries[0];
if (!entry) return;
const borderBoxSize = Array.isArray(entry.borderBoxSize)
? entry.borderBoxSize[0]
: (entry.borderBoxSize as unknown as ResizeObserverSize);
reportHeight(
borderBoxSize?.blockSize ?? dockRef.current?.offsetHeight ?? null,
);
},
visible,
);
if (!visible) return null;
return (
<div
className={`fixed bottom-0 left-0 z-30 border-t border-line bg-page ${
isResizing
? ""
: "transition-[right] duration-300 ease-[cubic-bezier(0.16,1,0.3,1)]"
}`}
ref={setDockRef}
className={classNames(
"fixed bottom-0 left-0 z-30 border-t border-line bg-page",
!isResizing &&
"transition-[right] duration-300 ease-[cubic-bezier(0.16,1,0.3,1)]",
)}
style={{ right: sidebarWidth }}
>
{hasPendingQuestions ? (

View file

@ -35,10 +35,11 @@ import {
formatDurationMs,
formatRelativeTime,
} from "../../lib/format";
import { classNames } from "../../lib/class-names";
import { useRunPullRequest } from "../../lib/queries";
import { sandboxRuntime } from "../../lib/run-sandbox-lifecycle";
import { ActionsMenu, type ActionsMenuProps } from "./actions";
import { classNames, type RunDetailRun } from "./model";
import type { RunDetailRun } from "./model";
export interface RunDetailHeaderActions {
approval: {
@ -326,6 +327,7 @@ function DurationPopover({
}) {
const endMs = completedAt != null ? Date.parse(completedAt) : now;
const sinceCreatedMs = Math.max(0, endMs - Date.parse(createdAt));
const isRunning = completedAt == null;
return (
<>
<PopoverHeader>Duration</PopoverHeader>
@ -335,8 +337,14 @@ function DurationPopover({
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(sinceCreatedMs)}</dd>
</div>
<div>
<dt className="text-fg-3">Active (inference + tools)</dt>
<dt className="text-fg-3">
Active (inference + tools){isRunning ? " — estimated" : ""}
</dt>
<dd className="mt-0.5 font-mono text-fg">{formatDurationMs(timing.active_time_ms)}</dd>
<dd className="mt-0.5 text-fg-3">
{formatDurationMs(timing.inference_time_ms)} inference ·{" "}
{formatDurationMs(timing.tool_time_ms)} tools
</dd>
</div>
</dl>
</>

View file

@ -1,6 +1,3 @@
import { useState } from "react";
import { useInterval } from "../../hooks/effects";
import {
isRunStatus,
mapRunToRunItem,
@ -8,16 +5,6 @@ import {
type Run,
} from "../../data/runs";
export function classNames(...classes: Array<string | false | null | undefined>) {
return classes.filter(Boolean).join(" ");
}
export function useTickingNow(intervalMs: number): number {
const [now, setNow] = useState(() => Date.now());
useInterval(() => setNow(Date.now()), intervalMs);
return now;
}
export type RunDetailRun = ReturnType<typeof mapRunToRunItem> & {
statusLabel: string;
statusDot: string;

View file

@ -1,7 +1,7 @@
import { Link, Outlet, type UIMatch } from "react-router";
import { classNames } from "../../lib/class-names";
import { sandboxTabVisible, type MaybeSandbox } from "../../lib/run-sandbox-lifecycle";
import { classNames } from "./model";
interface RunDetailTabDefinition {
name: string;

View file

@ -3,6 +3,7 @@ import { renderToStaticMarkup } from "react-dom/server";
import { StageState } from "@qltysh/fabro-api-client";
import type { Stage } from "../lib/stage-sidebar";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { StageChatView } from "./run-stages";
function stage(overrides: Partial<Stage> = {}): Stage {
@ -18,6 +19,7 @@ function stage(overrides: Partial<Stage> = {}): Stage {
resumedFromStageId: null,
startedAt: "2026-04-09T12:00:00Z",
providerUsed: null,
billing: makeBilledTokenCounts(),
...overrides,
};
}

View file

@ -1,9 +1,14 @@
import { describe, expect, test } from "bun:test";
import { renderToStaticMarkup } from "react-dom/server";
import type { ReasoningOutput } from "@qltysh/fabro-api-client";
import type {
BilledTokenCounts,
ReasoningOutput,
StageModelUsage,
} from "@qltysh/fabro-api-client";
import { EventDetails } from "./run-stages";
import { makeBilledTokenCounts } from "../lib/test-fixtures";
import { EventDetails, ModelUsagePopover } from "./run-stages";
const RUN_START = "2026-04-09T12:00:00Z";
@ -72,3 +77,77 @@ describe("EventDetails reasoning", () => {
expect(html).toContain(`${"x".repeat(280)}…`);
});
});
const PROVIDER_USED: StageModelUsage = {
mode: "agent",
provider: "moonshot",
model: "kimi-k3",
reasoning_effort: "max",
};
function popoverMarkup(counts: BilledTokenCounts): string {
return renderToStaticMarkup(
<ModelUsagePopover providerUsed={PROVIDER_USED} billing={counts} />,
);
}
describe("ModelUsagePopover billing", () => {
test("shows the visit's token buckets and cost next to the model", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 28_640,
output_tokens: 7_550,
reasoning_tokens: 1_200,
cache_read_tokens: 4_800,
cache_write_tokens: 1_500,
total_tokens: 43_690,
total_usd_micros: 720_000,
}),
);
expect(html).toContain("kimi-k3");
expect(html).toContain("Cache read");
expect(html).toContain("4.8k");
expect(html).toContain("Cache creation");
expect(html).toContain("1.5k");
expect(html).toContain("Uncached");
expect(html).toContain("28.6k");
// Output folds in reasoning tokens, matching the Billing tab.
expect(html).toContain("Output");
expect(html).toContain("8.8k");
expect(html).toContain("Cost");
expect(html).toContain("$0.72");
});
test("omits the token section for a stage that called no model", () => {
const html = popoverMarkup(makeBilledTokenCounts());
expect(html).toContain("kimi-k3");
expect(html).not.toContain("Tokens");
expect(html).not.toContain("Cost");
});
test("still shows tokens when nothing priced the stage", () => {
const html = popoverMarkup(
makeBilledTokenCounts({
input_tokens: 1_000,
output_tokens: 500,
total_tokens: 1_500,
}),
);
expect(html).toContain("Uncached");
expect(html).toContain("1.0k");
expect(html).not.toContain("Cost");
});
test("shows a provider-reported cost when token counts are unavailable", () => {
const html = popoverMarkup(
makeBilledTokenCounts({ total_usd_micros: 720_000 }),
);
expect(html).toContain("kimi-k3");
expect(html).toContain("Cost");
expect(html).toContain("$0.72");
});
});

View file

@ -67,7 +67,9 @@ import {
formatBytes,
formatDurationMs,
formatTokenCount,
formatUsdMicros,
} from "../lib/format";
import { billingTokenBuckets, hasBillingUsage } from "../lib/billing";
import { plural } from "../lib/plural";
import {
useRun,
@ -93,6 +95,7 @@ import {
type UnknownRecord,
} from "../lib/unknown";
import type {
BilledTokenCounts,
EventEnvelope,
ReasoningOutput,
StageHandler,
@ -866,10 +869,42 @@ export function formatStageModelUsageLabel(
return effort ? `${model}[${effort}]` : model;
}
function ModelUsagePopover({
const POPOVER_NUMBER = "block text-right font-mono tabular-nums";
/** Tokens and cost for this stage visit alone. */
function StageBillingRows({ billing }: { billing: BilledTokenCounts }) {
if (!hasBillingUsage(billing)) return null;
const buckets = billingTokenBuckets(billing);
const cost = formatUsdMicros(billing.total_usd_micros);
return (
<div className="mt-3">
<PopoverHeader>Tokens</PopoverHeader>
<PopoverRows>
{buckets.map((bucket) => (
<PopoverRow key={bucket.label} label={bucket.label}>
<span className={POPOVER_NUMBER}>
{bucket.value === 0
? "0"
: formatTokenCount(bucket.value, { compactDecimal: true })}
</span>
</PopoverRow>
))}
{cost && (
<PopoverRow label="Cost">
<span className={POPOVER_NUMBER}>{cost}</span>
</PopoverRow>
)}
</PopoverRows>
</div>
);
}
export function ModelUsagePopover({
providerUsed,
billing,
}: {
providerUsed: StageModelUsage;
billing: BilledTokenCounts;
}) {
return (
<>
@ -892,6 +927,7 @@ function ModelUsagePopover({
<PopoverRow label="Speed">{providerUsed.speed}</PopoverRow>
)}
</PopoverRows>
<StageBillingRows billing={billing} />
</>
);
}
@ -1905,6 +1941,7 @@ function EventsToolbar({
filteredCount,
totalCount,
providerUsed,
billing,
events,
runId,
stageId,
@ -1924,6 +1961,7 @@ function EventsToolbar({
filteredCount: number;
totalCount: number;
providerUsed: StageModelUsage | null;
billing: BilledTokenCounts;
events: EventEnvelope[];
runId: string;
stageId: string;
@ -2004,7 +2042,9 @@ function EventsToolbar({
className={`inline-flex items-center gap-1.5 text-xs text-fg-muted ${
showFilters ? "" : "ml-auto"
}`}
content={<ModelUsagePopover providerUsed={providerUsed} />}
content={
<ModelUsagePopover providerUsed={providerUsed} billing={billing} />
}
>
<CpuChipIcon className="size-3.5" aria-hidden="true" />
<span className="font-mono">{modelUsageLabel}</span>
@ -2359,6 +2399,7 @@ function RunStageActivityStage({
effectiveTab === "primary" ? turns.length : debugEvents.length
}
providerUsed={selectedStage.providerUsed}
billing={selectedStage.billing}
events={stageEventsQuery.data ?? []}
runId={runId}
stageId={selectedStageId}

View file

@ -26,6 +26,7 @@ import { ciConfig, columnForRun, columnStatusDisplay, columnStatuses, deriveCiSt
import type { CiStatus, CheckRun, CheckStatus, RunItem } from "../data/runs";
import { EmptyState } from "../components/state";
import { PullRequestChip } from "../components/pull-request-chip";
import { SizeChip } from "../components/size-chip";
import {
summarizeBatchLifecycleAction,
} from "../components/runs-list/batch-lifecycle";
@ -345,7 +346,7 @@ function PrCard({
// All inline footer metadata on PrCard belongs in this one row. Adding a new
// piece as a sibling `<div>` below the card body recreates a recurring bug
// where stats stack onto separate lines instead of sitting next to elapsed/actions.
// where stats stack onto separate lines instead of sitting next to size/actions.
function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
const hasActions = actions != null && actions.length > 0;
const hasStats =
@ -354,7 +355,7 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
(pr.additions != null && pr.additions !== 0) ||
(pr.deletions != null && pr.deletions !== 0);
if (!hasStats && !hasActions && pr.elapsed == null) return null;
if (!hasStats && !hasActions && pr.size == null) return null;
return (
<div className="mt-3 flex items-center gap-3 font-mono text-xs">
@ -416,9 +417,9 @@ function PrCardFooter({ pr, actions }: { pr: RunItem; actions?: string[] }) {
))}
</div>
)}
{pr.elapsed != null && (
<span className={`text-fg-muted ${hasActions ? "" : "ml-auto"}`}>
{pr.elapsed}
{pr.size != null && (
<span className={hasActions ? "inline-flex" : "ml-auto inline-flex"}>
<SizeChip size={pr.size} totalUsdMicros={pr.totalUsdMicros} />
</span>
)}
</div>

View file

@ -155,9 +155,16 @@ git diff --check
- Inference time is Fabro-observed LLM request/stream elapsed time, not
provider-reported model-only compute time.
- LLM retry backoff, queueing outside a request/stream, human waits, steering
waits, and scheduler gaps are wall time but not active time.
- Active timing is finalized-event based in v1; live active-time ticking can be
added later if it becomes necessary.
- Queueing outside a request/stream, human waits, steering waits, and scheduler
gaps are wall time but not active time. Retry delay inside an open LLM request
bracket follows the executor stopwatch and counts as inference time.
- ~~Active timing is finalized-event based in v1; live active-time ticking can
be added later if it becomes necessary.~~ **Superseded 2026-07-25.** It became
necessary: a run parked in one long agent stage reported ~12% of its wall time
as active, because in-flight stages contributed nothing. Stage projections now
accumulate inference and tool brackets from the event log and expose
`StageProjection::live_timing(now)`, the active-time twin of
`live_wall_time_ms`. Finalized values remain authoritative and still replace
the live estimate at terminal events. Implemented in PR #647.
- No compatibility layer is required for existing API clients or stored run
event data.

View file

@ -9597,6 +9597,11 @@ components:
type: ["string", "null"]
description: Optional contextual text shown alongside the question.
example: Latest draft
review_target:
description: Optional validated external resource that is the primary subject of this review question.
oneOf:
- $ref: "#/components/schemas/ReviewTarget"
- type: "null"
QuestionType:
description: The interaction type of a human-in-the-loop question.
@ -10673,7 +10678,38 @@ components:
- type: "null"
description: |
Per-attempt timing breakdown for the latest terminal attempt:
wall time plus the active inference/tool breakdown.
wall time plus the active inference/tool breakdown. Null while the
stage is still in flight; the live estimate is derived from
`live_inference_ms`, `live_tool_ms`, and any open bracket.
live_inference_ms:
type: integer
format: uint64
minimum: 0
default: 0
description: |
Inference time accumulated from closed brackets during the current
attempt. Live estimate only — the authoritative value arrives with
the terminal event and lands in `timing`. Excludes the currently
open bracket, whose span is measured from `inference.started_at`.
example: 78230
live_tool_ms:
type: integer
format: uint64
minimum: 0
default: 0
description: |
Tool time accumulated from closed tool batches during the current
attempt. A batch spans the first dispatched call through the
completion that drains the last outstanding one, so tools running
concurrently within a turn are counted once.
example: 7588
tool_batch:
oneOf:
- $ref: "#/components/schemas/StageToolBatchProjection"
- type: "null"
description: |
Open tool batch: when the batch started and which calls have not
yet reported completion.
usage:
$ref: "#/components/schemas/BilledTokenCounts"
model:
@ -10727,6 +10763,13 @@ components:
Open inference bracket, if the event log contains one. Present means
a model request was dispatched and no closing event has been seen —
not that the model is computing right now.
acp_started_at:
type: ["string", "null"]
format: date-time
description: >
Start of an external ACP agent process, if one is running. ACP
agents do not expose Fabro's internal LLM brackets, so the process
lifetime supplies their live inference estimate.
agent_control:
$ref: "#/components/schemas/AgentControlState"
description: Whether the agent is executing normally or waiting for steering after an interrupt.
@ -10734,6 +10777,37 @@ components:
$ref: "#/components/schemas/StageState"
description: Lifecycle state of the stage projection.
StageToolBatchProjection:
description: >
One open tool batch: tool calls dispatched together that have not all
reported completion. `open_call_ids` is a set rather than a count so a
duplicated completion in a replayed log cannot drain the batch early.
type: object
required:
- session_id
- started_at
- open_call_ids
properties:
session_id:
type: string
description: >
Root agent session that dispatched the batch. Transitions are
gated on it so delayed events from a replaced session cannot
mutate the current batch.
started_at:
type: string
format: date-time
description: >
When the batch opened — the first dispatched call observed while no
other calls were outstanding.
open_call_ids:
type: array
minItems: 1
uniqueItems: true
items:
type: string
description: Calls dispatched but not yet completed, by tool call id.
StageInferenceProjection:
description: >
One open inference bracket: a dispatched LLM request that has not yet
@ -11102,6 +11176,36 @@ components:
type: ["string", "null"]
description: Optional untrusted model-authored option preview captured for clients.
ReviewTargetKind:
description: The type of resource presented for human review.
type: string
enum:
- document
ReviewTarget:
description: A validated external resource presented as the primary subject of a human review question.
type: object
required:
- label
- url
- kind
properties:
label:
type: string
minLength: 1
maxLength: 200
description: Human-readable link label.
example: Quarry review exercise
url:
type: string
format: uri
minLength: 1
maxLength: 2048
description: Absolute HTTP or HTTPS URL opened by the reviewer.
example: https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef
kind:
$ref: "#/components/schemas/ReviewTargetKind"
InterviewQuestionRecord:
description: Storage shape of an interview question recorded in the event log.
type: object
@ -11131,6 +11235,10 @@ components:
format: double
context_display:
type: ["string", "null"]
review_target:
oneOf:
- $ref: "#/components/schemas/ReviewTarget"
- type: "null"
PendingInterviewRecord:
description: Pending interview question plus the time it entered the unresolved set.
@ -12244,9 +12352,17 @@ components:
observed LLM request/stream elapsed time; `tool_time_ms` is tool or
command execution elapsed time; `active_time_ms` equals
`inference_time_ms + tool_time_ms`.
For a terminal stage these come from the worker's own stopwatch and are
authoritative. For a stage still in flight they are a live estimate
reconstructed from the event log, and `active_time_ms` is clamped to
`wall_time_ms`. The estimate is replaced by the authoritative
breakdown when the stage reaches a terminal event.
type: object
required:
- wall_time_ms
- inference_time_ms
- tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@ -12278,9 +12394,16 @@ components:
Timing rollup for an entire run. Active fields sum work across stage
visits, so `active_time_ms` can exceed `wall_time_ms` when parallel
branches run concurrently.
For a running run, stages still in flight contribute a live estimate
rather than nothing, so wall and active both advance continuously.
Unlike `StageTiming`, active is not clamped to wall here — concurrent
branches can legitimately sum past run wall time.
type: object
required:
- wall_time_ms
- inference_time_ms
- tool_time_ms
- active_time_ms
properties:
wall_time_ms:
@ -12609,6 +12732,7 @@ components:
- status
- node_id
- visit
- billing
properties:
id:
$ref: "#/components/schemas/StageId"
@ -12668,6 +12792,15 @@ components:
format: date-time
description: Wall-clock time the latest attempt of this stage started, if known.
example: "2026-04-29T12:34:56Z"
billing:
$ref: "#/components/schemas/BilledTokenCounts"
description: >-
Token counts for this stage execution alone. `total_usd_micros` is
the provider-reported cost when there is one, otherwise the server
catalog's price for these tokens — the same pricing the
`/runs/{id}/billing` rows use. All-zero counts mean the stage made
no model calls. Unlike the billing rows, which sum every visit of a
node, this covers only this visit.
# ── File Diff Schemas ──────────────────────────────────────────────

View file

@ -40,6 +40,11 @@ Each handler type writes specific keys into the context after execution:
Agents can also emit arbitrary context updates by including a JSON object with a `context_updates` field in their response. See [Transitions](/workflows/transitions#agent-transitions).
The `review_target` key has an optional typed convention for human review
workflows. A human gate with `review_target=true` reads this exact flat key and
presents its document URL as the primary question link. See
[Review targets](/workflows/human-in-the-loop#review-targets).
### Command nodes
| Key | Value |

View file

@ -113,7 +113,7 @@ plan [label="Plan", prompt="Create an implementation plan."]
**Node identifiers** must start with a letter or underscore, followed by letters, digits, or underscores (e.g. `run_tests`, `gate_1`, `_private`).
Nodes referenced in edges are auto-created if not explicitly declared.
Every node used by an edge needs its own declaration. Validation fails when an edge names a node the workflow never declares, because that is nearly always a typo or a rename that missed an edge. The declaration can come before or after the edges that use it, and it can live in a subgraph.
### Edge declarations
@ -267,6 +267,7 @@ For the first node in each branch, `fidelity` resolves from the fork-to-branch e
| Attribute | Type | Description |
|---|---|---|
| `question_type` | String | Optional interview question type override: `yes_no`, `confirmation`, `multiple_choice`, `multi_select`, or `freeform`. Defaults to `freeform` when the gate only has a freeform edge; otherwise defaults to `multiple_choice`. |
| `review_target` | Boolean | When `true`, read and validate the typed `review_target` context value, then present it as the primary link in the question. Fabro generates the question text, so the node's `label` is not used. See [Review targets](/workflows/human-in-the-loop#review-targets). |
| `human.default_choice` | String | Target node to use when the question times out. |
### Manager loop (sub-workflow) nodes

View file

@ -60,6 +60,68 @@ confirm -> exit [label="[N] No"]
Supported values are `yes_no`, `confirmation`, `multiple_choice`, `multi_select`, and `freeform`.
### Review targets
A human gate can present one external document as the primary review link. Set
`review_target=true` on the gate:
```dot
review [
shape=hexagon,
review_target=true
]
review -> sync [label="[S] Review complete; sync the current Markdown"]
review -> address [label="[A] Ask the agent to address human feedback"]
review -> review [label="[C] Continue reviewing"]
```
Before the workflow reaches the gate, an agent, prompt, or command node must set
the flat `review_target` context key. A routing response can do this directly:
```json
{
"outcome": "succeeded",
"context_updates": {
"review_target": {
"label": "Quarry review exercise",
"url": "https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
"kind": "document"
}
}
}
```
Fabro then presents this question:
> Review the [Quarry review exercise](https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef) document, then choose the next action.
The link uses the target URL from context. The example uses a placeholder
secret, not a live Quarry document.
Fabro generates this question text from the target. A `label` on the gate is
not used while `review_target=true`.
The target object has three required fields:
| Field | Meaning |
|---|---|
| `label` | The link text. It must contain 1 to 200 characters and no control characters. |
| `url` | An absolute HTTP or HTTPS URL. It must contain a host, must not contain URL credentials, and must be at most 2048 characters. |
| `kind` | The resource type. The supported value is `document`. |
Fabro validates the target before it starts the interview. A missing or invalid
target fails the gate deterministically. A gate without `review_target=true`
ignores this context key and keeps its normal label.
The context value remains available after the human answers. A feedback loop
can return to the same gate without recreating the target. A later stage can
replace the target by writing a new value to the same context key.
Fabro does not fetch the URL. Web and Slack clients open it as an external link.
Treat bearer-capability links as secrets and only provide them to people who
can access the run.
### Default choice on timeout
If a human gate has a timeout configured, you can specify a default choice using the `human.default_choice` attribute:

View file

@ -438,6 +438,7 @@ fn api_question_to_question(question: &types::ApiQuestion) -> Question {
converted
.context_display
.clone_from(&question.context_display);
converted.review_target.clone_from(&question.review_target);
converted
}
@ -451,6 +452,9 @@ async fn ask_attach_question(question: Question, styles: &'static Styles) -> Ans
let rendered = styles.render_markdown(context_text);
eprint!("{rendered}");
}
if let Some(line) = fabro_interview::review_target_line(&question) {
eprintln!("{line}");
}
eprintln!("{} {}", styles.bold_cyan.apply_to("?"), question.text);
match question.question_type {

View file

@ -61,11 +61,8 @@ pub(crate) async fn create_run(
None
};
let mut validation = manifest_validation::validate_manifest(
&RunLayer::default(),
&built.manifest,
ctx.catalog()?,
)?;
let mut validation =
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
let diagnostics = api_diagnostics_to_local(&validation.workflow.diagnostics);
if !quiet {

View file

@ -110,7 +110,6 @@ pub(crate) async fn execute(
run_id,
run_spec.source_directory.as_deref(),
&run_dir,
Arc::clone(&catalog),
)
} else {
None
@ -237,13 +236,12 @@ fn build_fabro_run_tool_services(
current_run_id: RunId,
source_directory: Option<&str>,
run_dir: &Path,
catalog: Arc<Catalog>,
) -> Option<FabroRunToolServices> {
if worker_token.trim().is_empty() {
return None;
}
let backend = ClientBackend::new(Arc::new(client))
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder { catalog }));
.with_manifest_builder(Arc::new(WorkerRunManifestBuilder));
Some(FabroRunToolServices {
backend: Arc::new(backend),
current_run_id,
@ -252,9 +250,7 @@ fn build_fabro_run_tool_services(
})
}
struct WorkerRunManifestBuilder {
catalog: Arc<Catalog>,
}
struct WorkerRunManifestBuilder;
impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
fn build_run_manifest(
@ -263,12 +259,7 @@ impl fabro_tool::RunManifestBuilder for WorkerRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> fabro_tool::ToolResult<RunManifest> {
run_tool_manifest::build_run_tool_manifest(
spec,
cwd,
user_settings_path,
Arc::clone(&self.catalog),
)
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}
@ -1361,6 +1352,7 @@ mod tests {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
})),
Some(WorkerTitlePhase::Waiting)
);

View file

@ -23,11 +23,7 @@ pub(crate) fn run(
user_settings_path: Some(active_settings_path(None)),
..Default::default()
})?;
let response = manifest_validation::validate_manifest(
&RunLayer::default(),
&built.manifest,
base_ctx.catalog()?,
)?;
let response = manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)?;
let diagnostics = api_diagnostics_to_local(&response.workflow.diagnostics);
if base_ctx.json_output() {

View file

@ -49,54 +49,49 @@ where
pub(crate) fn print_diagnostics(diagnostics: &[Diagnostic], styles: &Styles, printer: Printer) {
for d in diagnostics {
let location = match (&d.node_id, &d.edge) {
(Some(node), _) => format!(" [node: {node}]"),
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
_ => String::new(),
};
let source_prefix = source_prefix(d);
match d.severity {
Severity::Error if source_prefix.is_empty() => fabro_util::printerr!(
printer,
"{}{location}: {} ({})",
styles.red.apply_to("error"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Error => fabro_util::printerr!(
printer,
"{}: {source_prefix}{}{location} ({})",
styles.red.apply_to("error"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Warning if source_prefix.is_empty() => fabro_util::printerr!(
printer,
"{}{location}: {} ({})",
styles.yellow.apply_to("warning"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Warning => fabro_util::printerr!(
printer,
"{}: {source_prefix}{}{location} ({})",
styles.yellow.apply_to("warning"),
d.message,
styles.dim.apply_to(&d.rule),
),
Severity::Info => fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(if source_prefix.is_empty() {
format!("info{location}: {} ({})", d.message, d.rule)
} else {
format!("info: {source_prefix}{}{location} ({})", d.message, d.rule)
}),
),
print_diagnostic(d, styles, printer);
// The fix is the actionable half of a diagnostic, so it follows every
// severity rather than hiding behind --verbose. Rules that have nothing
// useful to suggest leave it unset.
if let Some(fix) = &d.fix {
fabro_util::printerr!(printer, " {} {fix}", styles.dim.apply_to("fix:"));
}
}
}
fn print_diagnostic(d: &Diagnostic, styles: &Styles, printer: Printer) {
let location = match (&d.node_id, &d.edge) {
(Some(node), _) => format!(" [node: {node}]"),
(_, Some((from, to))) => format!(" [edge: {from} -> {to}]"),
_ => String::new(),
};
let source_prefix = source_prefix(d);
let body = if source_prefix.is_empty() {
format!("{location}: {}", d.message)
} else {
format!(": {source_prefix}{}{location}", d.message)
};
match d.severity {
Severity::Error => fabro_util::printerr!(
printer,
"{}{body} ({})",
styles.red.apply_to("error"),
styles.dim.apply_to(&d.rule),
),
Severity::Warning => fabro_util::printerr!(
printer,
"{}{body} ({})",
styles.yellow.apply_to("warning"),
styles.dim.apply_to(&d.rule),
),
Severity::Info => fabro_util::printerr!(
printer,
"{}",
styles.dim.apply_to(format!("info{body} ({})", d.rule)),
),
}
}
fn source_prefix(diagnostic: &Diagnostic) -> String {
match (
diagnostic.source_path.as_deref(),

View file

@ -101,6 +101,40 @@ fn create_uses_explicit_server_target_and_prints_remote_run_id() {
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
}
#[test]
fn create_defers_provider_validation_to_the_server() {
let context = test_context!();
let server = MockServer::start();
let run_id = unique_run_id();
let mock = server.mock(|when, then| {
when.method("POST")
.path("/api/v1/runs")
.body_includes(r#"provider=\"server-only\""#);
then.status(201)
.header("Content-Type", "application/json")
.body(run_status_response(run_id.as_str(), "submitted").to_string());
});
let output = context
.create_cmd()
.args([
"--server",
&format!("{}/api/v1", server.base_url()),
"--dry-run",
fixture("server-model.fabro").to_str().unwrap(),
])
.output()
.expect("command should execute");
assert!(
output.status.success(),
"local validation should not reject a server-owned provider\nstdout:\n{}\nstderr:\n{}",
String::from_utf8_lossy(&output.stdout),
String::from_utf8_lossy(&output.stderr)
);
mock.assert();
assert_eq!(output_stdout(&output).trim(), run_id.as_str());
}
#[test]
fn create_uses_configured_server_target_without_server_flag() {
let context = test_context!();

View file

@ -95,7 +95,9 @@ fn graph_allow_invalid_renders_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
");
let svg = read_text(&output_path);
@ -119,7 +121,9 @@ fn graph_invalid_workflow_fails_after_diagnostics() {
----- stdout -----
----- stderr -----
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}

View file

@ -52,7 +52,9 @@ fn preflight_invalid_workflow_fails_with_validation_output() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}
@ -74,7 +76,9 @@ fn preflight_rejects_unbound_template_inputs() {
Goal: Demo
error: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
error: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
× Validation failed
");
}

View file

@ -69,6 +69,24 @@ fn simple_does_not_connect_to_configured_server() {
);
}
/// Offline validation has no model catalog, so a model or provider the server
/// owns must pass rather than be reported as unknown.
#[test]
fn server_owned_provider_is_not_rejected_by_offline_validation() {
let context = test_context!();
let mut cmd = context.validate();
cmd.arg(fixture("server-model.fabro"));
fabro_snapshot!(context.filters(), cmd, @"
success: true
exit_code: 0
----- stdout -----
----- stderr -----
Workflow: ServerModel (3 nodes, 2 edges)
Graph: [FIXTURES]/server-model.fabro
Validation: OK
");
}
#[test]
fn branching() {
let context = test_context!();
@ -82,6 +100,7 @@ fn branching() {
Workflow: Branch (6 nodes, 6 edges)
Graph: [FIXTURES]/branching.fabro
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
fix: Add retry_target or fallback_retry_target attribute
Validation: OK
");
}
@ -163,7 +182,9 @@ fn bare_fabro_with_unbound_inputs_validates_structurally_with_warning() {
Workflow: TemplatedUnbound (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound.fabro
warning: [FIXTURES]/templated_unbound.fabro:2:26: undefined template variable `inputs.app_dir` in graph attribute `goal` (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
warning: [FIXTURES]/templated_unbound.fabro:7:44: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
Validation: OK
");
}
@ -186,6 +207,7 @@ fn bare_fabro_with_unbound_inputs_in_imported_prompt_validates_structurally_with
Workflow: TemplatedUnboundImported (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_imported/workflow.fabro
warning: [FIXTURES]/templated_unbound_imported/work.md:1:12: undefined template variable `inputs.app_dir` in node `work` attribute `prompt` [node: work] (template_undefined_variable)
fix: bind `inputs.app_dir` via `[run.inputs]` in workflow.toml, or pass `--input inputs.app_dir=<value>`
Validation: OK
");
}
@ -207,6 +229,7 @@ fn bare_fabro_with_unbound_inputs_in_template_partial_validates_structurally_wit
Workflow: TemplatedUnboundPartial (3 nodes, 2 edges)
Graph: [FIXTURES]/templated_unbound_partial/workflow.fabro
warning: [FIXTURES]/templated_unbound_partial/test-include.partial.md:1:4: undefined template variable `inputs.hello` in node `test_imported_include` attribute `prompt` [node: test_imported_include] (template_undefined_variable)
fix: bind `inputs.hello` via `[run.inputs]` in workflow.toml, or pass `--input inputs.hello=<value>`
Validation: OK
");
}
@ -278,6 +301,26 @@ fn validate_reports_missing_template_dependency() {
");
}
/// A node named only by an edge is almost always a typo, so validation must
/// fail instead of quietly running it as a default agent stage.
#[test]
fn edge_only_node() {
let context = test_context!();
let mut cmd = context.validate();
cmd.arg(fixture("edge_only_node.fabro"));
fabro_snapshot!(context.filters(), cmd, @"
success: false
exit_code: 1
----- stdout -----
----- stderr -----
Workflow: EdgeOnlyNode (2 nodes, 2 edges)
Graph: [FIXTURES]/edge_only_node.fabro
error [node: misspelled_node]: Node 'misspelled_node' is referenced by edge 'start -> misspelled_node' but has no node declaration (edge_target_exists)
fix: Declare node 'misspelled_node' or correct the edge endpoint
× Validation failed
");
}
#[test]
fn invalid() {
let context = test_context!();
@ -291,7 +334,9 @@ fn invalid() {
Workflow: Invalid (2 nodes, 1 edges)
Graph: [FIXTURES]/invalid.fabro
error: Pipeline must have exactly one start node (shape=Mdiamond or id start/Start) (start_node)
fix: Add a node with shape=Mdiamond or id 'start'
error [node: exit]: Exit node 'exit' has 1 outgoing edge(s) but must have none (exit_no_outgoing)
fix: Remove outgoing edges from the exit node
× Validation failed
");
}

View file

@ -19,6 +19,7 @@ fn dry_run_branching() {
Goal: Implement and validate a feature
warning [node: implement]: Node 'implement' has goal_gate=true but no retry_target or fallback_retry_target (goal_gate_has_retry)
fix: Add retry_target or fallback_retry_target attribute
Run: [ULID]
Web UI: http://localhost:3000/runs/[ULID]
Sandbox: local (ready in [TIME])

View file

@ -1,11 +1,8 @@
use std::path::Path;
use std::sync::Arc;
use fabro_api::types;
use fabro_config::load_llm_catalog_settings;
use fabro_model::Catalog;
use fabro_server::run_tool_manifest;
use fabro_tool::{RunManifestBuilder, ToolError, ToolResult, ValidatedCreateRunSpec};
use fabro_tool::{RunManifestBuilder, ToolResult, ValidatedCreateRunSpec};
#[derive(Default)]
pub(crate) struct McpRunManifestBuilder;
@ -17,20 +14,6 @@ impl RunManifestBuilder for McpRunManifestBuilder {
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult<types::RunManifest> {
build_mcp_run_manifest(spec, cwd, user_settings_path)
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path)
}
}
fn build_mcp_run_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
) -> ToolResult<types::RunManifest> {
let llm_catalog_settings = load_llm_catalog_settings(Some(user_settings_path))
.map_err(|err| ToolError::message(err.to_string()))?;
let catalog = Arc::new(
Catalog::from_builtin_with_overrides(&llm_catalog_settings)
.map_err(|err| ToolError::message(err.to_string()))?,
);
run_tool_manifest::build_run_tool_manifest(spec, cwd, user_settings_path, catalog)
}

View file

@ -1134,6 +1134,7 @@ mod runs {
id: stage_id.clone(),
name: name.to_owned(),
handler,
billing: BilledTokenCounts::default(),
status,
wall_time_ms,
node_id: stage_id.node_id().to_owned(),
@ -1752,6 +1753,7 @@ mod runs {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
},
ApiQuestion {
id: "q-002".into(),
@ -1775,6 +1777,7 @@ mod runs {
allow_freeform: true,
timeout_seconds: None,
context_display: None,
review_target: None,
},
]
}

View file

@ -1,41 +1,30 @@
use std::collections::HashMap;
use std::sync::Arc;
use anyhow::Result;
use fabro_api::types;
use fabro_config::{EnvironmentLayer, MergeMap, RunLayer};
use fabro_model::Catalog;
use fabro_config::RunLayer;
use fabro_workflow::pipeline::TEMPLATE_UNDEFINED_VARIABLE_RULE;
use crate::run_manifest;
/// Validate a manifest without a model catalog.
///
/// Every caller is a client — the CLI, an MCP server, a run worker — and a
/// client's catalog is its own, not the server's. Judging model and provider
/// availability here would reject workflows the server can run, so that is
/// left to the server on create.
pub fn validate_manifest(
manifest_run_defaults: &RunLayer,
manifest: &types::RunManifest,
catalog: Arc<Catalog>,
) -> Result<types::ValidateResponse> {
validate_manifest_with_environment_defaults(
manifest_run_defaults,
&fabro_environment::seeded_catalog_layer(),
manifest,
catalog,
)
}
pub fn validate_manifest_with_environment_defaults(
manifest_run_defaults: &RunLayer,
manifest_environment_defaults: &MergeMap<EnvironmentLayer>,
manifest: &types::RunManifest,
catalog: Arc<Catalog>,
) -> Result<types::ValidateResponse> {
let prepared = run_manifest::prepare_manifest_with_environment_defaults(
manifest_run_defaults,
manifest_environment_defaults,
&fabro_environment::seeded_catalog_layer(),
&HashMap::new(),
manifest,
)?;
let validated =
run_manifest::validate_prepared_manifest(&prepared, catalog).map_err(anyhow::Error::new)?;
let validated = run_manifest::validate_prepared_manifest_structural(&prepared)
.map_err(anyhow::Error::new)?;
Ok(run_manifest::validate_response(&prepared, &validated))
}

View file

@ -1217,7 +1217,7 @@ async fn reconnect_run_sandbox(
.await
.map_err(|err| ApiError::new(StatusCode::CONFLICT, err.to_string()))?;
sandbox
.start()
.activate()
.await
.map_err(|err| ApiError::new(StatusCode::CONFLICT, err.display_with_causes()))?;
Ok(sandbox)

View file

@ -35,7 +35,8 @@ use fabro_util::check_report::{CheckDetail, CheckReport, CheckResult, CheckSecti
use fabro_validate::Severity;
use fabro_workflow::Error as WorkflowError;
use fabro_workflow::operations::{
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_ready_providers,
CreateRunInput, ValidateInput, WorkflowInput, validate, validate_with_catalog,
validate_with_ready_providers,
};
use fabro_workflow::pipeline::Validated;
use fabro_workflow::run_materialization::materialize_run_with_ready_providers;
@ -192,12 +193,18 @@ pub(crate) fn validate_prepared_manifest(
validate_prepared_manifest_with_vars(prepared, catalog, HashMap::new())
}
pub(crate) fn validate_prepared_manifest_structural(
prepared: &PreparedManifest,
) -> Result<Validated, WorkflowError> {
validate(manifest_validate_input(prepared, HashMap::new()))
}
pub(crate) fn validate_prepared_manifest_with_vars(
prepared: &PreparedManifest,
catalog: Arc<Catalog>,
vars: HashMap<String, String>,
) -> Result<Validated, WorkflowError> {
validate(manifest_validate_input(prepared, catalog, vars))
validate_with_catalog(manifest_validate_input(prepared, vars), catalog)
}
pub(crate) fn validate_prepared_manifest_for_preflight(
@ -207,14 +214,14 @@ pub(crate) fn validate_prepared_manifest_for_preflight(
ready_providers: &[ProviderId],
) -> Result<Validated, WorkflowError> {
validate_with_ready_providers(
manifest_validate_input(prepared, catalog, vars),
manifest_validate_input(prepared, vars),
catalog,
ready_providers,
)
}
fn manifest_validate_input(
prepared: &PreparedManifest,
catalog: Arc<Catalog>,
vars: HashMap<String, String>,
) -> ValidateInput {
ValidateInput {
@ -223,7 +230,6 @@ fn manifest_validate_input(
vars,
cwd: prepared.cwd.clone(),
custom_transforms: Vec::new(),
catalog,
}
}

View file

@ -1,20 +1,22 @@
use std::path::{Path, PathBuf};
use std::sync::Arc;
use fabro_api::types;
use fabro_config::{CliLayer, RunGoalLayer, RunLayer};
use fabro_manifest::{ManifestBuildInput, RunOverrideInput};
use fabro_model::Catalog;
use fabro_tool::{ToolError, ToolResult, ValidatedCreateRunSpec};
use fabro_types::settings::interp::InterpString;
use crate::manifest_validation;
/// Build and validate a run manifest for the `fabro_run_create` tool.
///
/// Validation is structural. The caller is a client — an MCP server or a run
/// worker — whose catalog is its own, not the server's, so judging model and
/// provider availability here would reject workflows the server can run.
pub fn build_run_tool_manifest(
spec: &ValidatedCreateRunSpec,
cwd: &Path,
user_settings_path: &Path,
catalog: Arc<Catalog>,
) -> ToolResult<types::RunManifest> {
let built = fabro_manifest::build_run_manifest(ManifestBuildInput {
workflow: PathBuf::from(&spec.workflow),
@ -30,7 +32,7 @@ pub fn build_run_tool_manifest(
.map_err(|err| ToolError::from_anyhow(&err))?;
let mut validation =
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest, catalog)
manifest_validation::validate_manifest(&RunLayer::default(), &built.manifest)
.map_err(|err| ToolError::from_anyhow(&err))?;
manifest_validation::promote_template_undefined_variables_to_errors(&mut validation);
if !validation.ok {
@ -103,6 +105,63 @@ mod tests {
use super::*;
fn create_run_spec(workflow: &str) -> ValidatedCreateRunSpec {
ValidatedCreateRunSpec::try_from(CreateRunSpec {
workflow: workflow.to_string(),
run_id: None,
parent_id: None,
cwd: None,
goal: None,
goal_file: None,
inputs: HashMap::new(),
labels: HashMap::new(),
model: None,
provider: None,
environment: None,
dry_run: None,
auto_approve: None,
preserve_sandbox: None,
start: None,
})
.expect("create spec should validate")
}
/// The tool runs on a client, whose catalog is not the server's, so a
/// server-owned model must reach the server rather than fail here.
#[expect(
clippy::disallowed_methods,
reason = "sync test writes one workflow fixture before building the manifest"
)]
#[test]
fn server_owned_provider_is_not_rejected_by_tool_manifest_validation() {
let dir = tempfile::tempdir().expect("temp dir should be created");
let workflow = dir.path().join("server-model.fabro");
std::fs::write(
&workflow,
r#"digraph ServerModel {
graph [goal="Use a server-owned model"]
start [shape=Mdiamond]
work [prompt="Do work", model="private-model", provider="server-only"]
exit [shape=Msquare]
start -> work -> exit
}"#,
)
.expect("workflow fixture should be written");
let manifest = build_run_tool_manifest(
&create_run_spec(&workflow.to_string_lossy()),
dir.path(),
&dir.path().join("settings.toml"),
)
.expect("tool validation should leave provider availability to the server");
let encoded = serde_json::to_string(&manifest).expect("manifest should serialize");
assert!(
encoded.contains("server-only"),
"the authored provider should survive into the manifest: {encoded}"
);
}
#[test]
fn manifest_args_preserve_input_provenance() {
let spec = ValidatedCreateRunSpec::try_from(CreateRunSpec {

View file

@ -719,6 +719,7 @@ impl SlackService {
allow_freeform: props.allow_freeform,
timeout_seconds: props.timeout_seconds,
context_display: props.context_display.clone(),
review_target: props.review_target.clone(),
});
let blocks = slack_blocks::question_to_blocks(
&event.run_id.to_string(),
@ -3708,6 +3709,7 @@ fn runtime_question_from_interview_record(question: &InterviewQuestionRecord) ->
stage: question.stage.clone(),
metadata: HashMap::new(),
context_display: question.context_display.clone(),
review_target: question.review_target.clone(),
}
}
@ -3721,6 +3723,7 @@ fn api_question_from_interview_record(question: &InterviewQuestionRecord) -> Api
allow_freeform: question.allow_freeform,
timeout_seconds: question.timeout_seconds,
context_display: question.context_display.clone(),
review_target: question.review_target.clone(),
}
}

View file

@ -2,6 +2,7 @@ use std::collections::HashMap;
use std::sync::Arc;
use chrono::{DateTime, Utc};
use fabro_model::Catalog;
use fabro_types::{
Graph, RunProjection, StageHandler, StageId, StageProjection, StageState, StageTiming,
};
@ -22,6 +23,7 @@ fn run_stage_from_projection(
stage_id: &StageId,
stage: &StageProjection,
graph: &Graph,
catalog: &Catalog,
now: DateTime<Utc>,
) -> RunStage {
let handler = stage.handler.unwrap_or_else(|| {
@ -36,6 +38,7 @@ fn run_stage_from_projection(
id: stage_id.clone(),
name: stage_id.node_id().to_owned(),
handler,
billing: stage.billed_usage(Some(catalog)).into_owned(),
status: stage.effective_state(),
wall_time_ms: stage.live_wall_time_ms(now),
node_id: stage_id.node_id().to_owned(),
@ -67,9 +70,10 @@ async fn list_run_stages(
let now = Utc::now();
let graph = projection.spec().graph();
let catalog = state.catalog();
let stages = projection
.iter_stages()
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, now))
.map(|(stage_id, stage)| run_stage_from_projection(stage_id, stage, graph, &catalog, now))
.collect::<Vec<_>>();
(StatusCode::OK, Json(ListResponse::new(stages))).into_response()
@ -175,7 +179,7 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
index
});
let row = &mut rows[index];
let stage_timing = billing_stage_timing(stage, now);
let stage_timing = stage.live_timing(now);
row.timing = row.timing.saturating_add(&stage_timing);
if stage_id.visit() >= row.latest_visit {
@ -188,20 +192,6 @@ fn live_billing_rows(projection: &RunProjection, now: DateTime<Utc>) -> Vec<Live
rows
}
/// Per-visit timing for a stage. For terminal visits, the stored breakdown is
/// used directly. For in-flight visits, fall back to the live wall-clock since
/// `started_at` (no active breakdown yet — that is only finalized at terminal
/// event time in v1).
fn billing_stage_timing(stage: &StageProjection, now: DateTime<Utc>) -> StageTiming {
if let Some(timing) = stage.timing {
return timing;
}
if let Some(live_wall) = stage.live_wall_time_ms(now) {
return StageTiming::wall_only(live_wall);
}
StageTiming::default()
}
fn stage_has_billing_row(stage: &StageProjection) -> bool {
stage.completion.is_some()
|| stage.timing.is_some()

View file

@ -11,7 +11,7 @@ use axum_extra::extract::Query as ExtraQuery;
use base64::Engine as _;
use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
use bytes::Bytes;
use chrono::Utc;
use chrono::{DateTime, Utc};
use fabro_api::types::{
BoardColumn, RunManifest, SubmitAnswerRequest, UpdateRunParentRequest, UpdateRunRequest,
};
@ -23,7 +23,7 @@ use fabro_store::{
};
use fabro_types::settings::ResolveError;
use fabro_types::{
AutomationRef, Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
AutomationRef, Principal, Run, RunClientProvenance, RunId, RunProvenance, RunServerProvenance,
RunStatusKind, StageContextWindow, StageContextWindowStaleness,
StageContextWindowUnavailableReason, StageHandler, StageModelUsage, StageProjection,
SystemActorKind, WorkflowSettings, parse_blob_ref,
@ -298,7 +298,7 @@ async fn validate_parent_link(
}
async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
match state.stores.run_summaries.get(run_id, Utc::now()).await {
match run_summary_at(state, run_id, Utc::now()).await {
Ok(Some(summary)) => (
StatusCode::OK,
Json(state.decorate_run_summary(summary).await),
@ -311,6 +311,28 @@ async fn updated_run_response(state: &AppState, run_id: &RunId) -> Response {
}
}
/// Read the durable summary and overlay its timing from the live projection.
///
/// The SQLite read model stores active timing as of the most recent event.
/// An open inference or tool bracket keeps accruing between events, so detail
/// reads need the projection's current estimate while the run is non-terminal.
async fn run_summary_at(
state: &AppState,
run_id: &RunId,
now: DateTime<Utc>,
) -> fabro_store::Result<Option<Run>> {
let Some(mut summary) = state.stores.run_summaries.get(run_id, now).await? else {
return Ok(None);
};
if summary.timestamps.completed_at.is_none() {
let projection = state.stores.runs.get_cached_projection(run_id).await?;
if let Some(timing) = projection.and_then(|projection| projection.live_run_timing(now)) {
summary.timing = Some(timing);
}
}
Ok(Some(summary))
}
async fn list_runs(
_auth: RequiredRunManagementActor,
State(state): State<Arc<AppState>>,
@ -934,7 +956,7 @@ async fn get_run_status(
RequireRunManagementTarget(id, _actor): RequireRunManagementTarget,
State(state): State<Arc<AppState>>,
) -> Response {
match state.stores.run_summaries.get(&id, Utc::now()).await {
match run_summary_at(&state, &id, Utc::now()).await {
Ok(Some(run)) => {
(StatusCode::OK, Json(state.decorate_run_summary(run).await)).into_response()
}

View file

@ -881,7 +881,7 @@ async fn reconnect_run_sandbox_instance(
let detail = render_with_causes(&err.to_string(), &collect_causes(err.as_ref()));
ApiError::new(StatusCode::CONFLICT, detail).into_response()
})?;
sandbox.start().await.map_err(|err| {
sandbox.activate().await.map_err(|err| {
ApiError::new(StatusCode::CONFLICT, err.display_with_causes()).into_response()
})?;
Ok(sandbox)
@ -927,7 +927,7 @@ async fn reconnect_daytona_sandbox_instance(
.map_err(|err| {
ApiError::new(StatusCode::CONFLICT, err.display_with_causes()).into_response()
})?;
sandbox.start().await.map_err(|err| {
sandbox.activate().await.map_err(|err| {
ApiError::new(StatusCode::CONFLICT, err.display_with_causes()).into_response()
})?;
Ok(sandbox)

View file

@ -730,6 +730,10 @@ async fn build_agent_session(
let sandbox = reconnect_for_run(sandbox_instance, daytona_api_key, Some(run_id))
.await
.map_err(AskFabroBuildError::SandboxUnavailable)?;
sandbox
.activate()
.await
.map_err(|err| AskFabroBuildError::SandboxUnavailable(anyhow::Error::new(err)))?;
let sandbox: Arc<dyn fabro_agent::Sandbox> = Arc::from(sandbox);
// No optional web-tool dependencies: `AskFabroToolAccessPolicy` denies
// `web_search` and `web_fetch`, and both `tools()` and the prompt are

View file

@ -4765,6 +4765,7 @@ channel = "#deploys"
allow_freeform: true,
timeout_seconds: None,
context_display: None,
review_target: None,
},
)
.await;
@ -5259,6 +5260,64 @@ fn test_billed_usage(
.unwrap()
}
async fn create_billed_retry_run(state: &Arc<AppState>, run_id: RunId) {
create_durable_run_with_events(state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
state,
run_id,
"verify",
1,
&workflow_event::Event::StageFailed {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(1200),
billing: Some(test_billed_usage("gpt-old", 100, 10)),
actor: None,
},
)
.await;
append_scoped_stage_event(
state,
run_id,
"verify",
2,
&workflow_event::Event::StageCompleted {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
timing: fabro_types::StageTiming::wall_only(800),
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing: Some(test_billed_usage("gpt-new", 200, 20)),
failure: None,
notes: None,
files_touched: Vec::new(),
context_updates: None,
jump_to_node: None,
context_values: None,
node_visits: None,
loop_failure_signatures: None,
restart_failure_signatures: None,
response: None,
attempt: 2,
max_attempts: 2,
},
)
.await;
}
#[tokio::test]
async fn list_run_stages_distinguishes_visits() {
let state = test_app_state_with_isolated_storage();
@ -5503,6 +5562,50 @@ async fn list_run_stages_exposes_execution_identity_for_resumed_stage() {
assert_eq!(second["resumed_from_stage_id"], "work@1");
}
#[tokio::test]
async fn run_billing_includes_live_stage_timing_in_rows_and_totals() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
workflow_run_started_event(run_id),
])
.await;
append_scoped_stage_event(
&state,
run_id,
"work",
1,
&stage_started_event("work", "command"),
)
.await;
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/billing")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let stages = body["stages"].as_array().unwrap();
assert_eq!(stages.len(), 1);
let row_timing = &stages[0]["timing"];
assert!(row_timing["active_time_ms"].as_u64().unwrap() > 0);
assert_eq!(row_timing["tool_time_ms"], row_timing["active_time_ms"]);
assert_eq!(&body["totals"]["timing"], row_timing);
}
/// `checkpoint.completed_nodes` records every visit, so a looped node appears
/// once per re-entry. Billing must dedup so a retried node renders as one row
/// and `runtime_secs` is summed across all visits exactly once.
@ -5654,65 +5757,9 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
let failed_usage = test_billed_usage("gpt-old", 100, 10);
create_billed_retry_run(&state, run_id).await;
let success_usage = test_billed_usage("gpt-new", 200, 20);
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
1,
&workflow_event::Event::StageFailed {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
failure: FailureDetail::new("try again", FailureCategory::TransientInfra),
will_retry: true,
timing: fabro_types::StageTiming::wall_only(1200),
billing: Some(failed_usage),
actor: None,
},
)
.await;
append_scoped_stage_event(
&state,
run_id,
"verify",
2,
&workflow_event::Event::StageCompleted {
node_id: "verify".to_string(),
name: "Verify".to_string(),
index: 1,
timing: fabro_types::StageTiming::wall_only(800),
status: "succeeded".to_string(),
preferred_label: None,
suggested_next_ids: Vec::new(),
billing: Some(success_usage.clone()),
failure: None,
notes: None,
files_touched: Vec::new(),
context_updates: None,
jump_to_node: None,
context_values: None,
node_visits: None,
loop_failure_signatures: None,
restart_failure_signatures: None,
response: None,
attempt: 2,
max_attempts: 2,
},
)
.await;
let mut latest_outcome: Outcome<Option<fabro_model::BilledModelUsage>> = Outcome::success();
latest_outcome.usage = Some(success_usage);
latest_outcome.timing = Some(fabro_types::StageTiming::wall_only(800));
@ -5746,7 +5793,6 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
.unwrap();
let response = app
.clone()
.oneshot(
Request::builder()
.method("GET")
@ -5791,6 +5837,76 @@ async fn run_billing_sums_usage_across_retry_visits_and_uses_latest_model() {
assert_eq!(new_model["billing"]["input_tokens"], 200);
}
/// The stage popover reads `billing` off the stages list, so it must be scoped
/// to one visit — unlike the Billing tab's rows, which sum every visit of a
/// node.
#[tokio::test]
async fn list_run_stages_reports_billing_per_visit() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_billed_retry_run(&state, run_id).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let first = stage_entry(&body, "verify@1");
assert_eq!(first["billing"]["input_tokens"], 100);
assert_eq!(first["billing"]["output_tokens"], 10);
assert_eq!(first["billing"]["total_usd_micros"], 110);
let second = stage_entry(&body, "verify@2");
assert_eq!(second["billing"]["input_tokens"], 200);
assert_eq!(second["billing"]["output_tokens"], 20);
assert_eq!(second["billing"]["total_usd_micros"], 220);
}
#[tokio::test]
async fn list_run_stages_reports_zero_billing_for_a_stage_that_called_no_model() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
])
.await;
let started = stage_started_event("script", "command");
append_scoped_stage_event(&state, run_id, "script", 1, &started).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}/stages")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let billing = &stage_entry(&body, "script@1")["billing"];
assert_eq!(billing["input_tokens"], 0);
assert_eq!(billing["output_tokens"], 0);
// No model ran, so there is nothing to price — not a $0.00 cost.
assert!(billing.get("total_usd_micros").is_none());
}
#[tokio::test]
async fn list_run_stages_shows_retrying_after_failed_event() {
let state = test_app_state_with_isolated_storage();
@ -7800,6 +7916,51 @@ async fn get_run_status_returns_status() {
assert!(body["labels"].is_object());
}
#[tokio::test]
async fn get_run_status_advances_live_active_timing_between_events() {
let state = test_app_state_with_isolated_storage();
let app = crate::test_support::build_test_router(Arc::clone(&state));
let run_id = RunId::new();
create_durable_run_with_events(&state, run_id, &[
workflow_event::Event::RunSubmitted {
definition_blob: None,
},
workflow_event::Event::RunStarting,
workflow_event::Event::RunRunning,
workflow_run_started_event(run_id),
])
.await;
append_scoped_stage_event(
&state,
run_id,
"work",
1,
&stage_started_event("work", "command"),
)
.await;
// The SQLite summary stores timing at the StageStarted event. A later
// detail read must overlay the in-flight command's active time from the
// projection even though no newer event has arrived.
tokio::time::sleep(std::time::Duration::from_millis(20)).await;
let response = app
.oneshot(
Request::builder()
.method("GET")
.uri(api(&format!("/runs/{run_id}")))
.body(Body::empty())
.unwrap(),
)
.await
.unwrap();
let body = response_json!(response, StatusCode::OK).await;
let timing = &body["timing"];
assert!(timing["active_time_ms"].as_u64().unwrap() > 0);
assert_eq!(timing["tool_time_ms"], timing["active_time_ms"]);
assert!(timing["wall_time_ms"].as_u64().unwrap() >= timing["active_time_ms"].as_u64().unwrap());
}
#[tokio::test]
async fn get_run_status_not_found() {
let app = test_app_with();
@ -8031,6 +8192,7 @@ async fn submit_pending_interview_answer_rejects_invalid_answer_shape() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
},
};
@ -8059,6 +8221,7 @@ fn validate_answer_for_question_accepts_no_for_confirmation() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
};
let result = validate_answer_for_question(&question, &Answer::no());
@ -8077,6 +8240,7 @@ fn answer_from_typed_yes_request_maps_to_yes_answer() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
};
let req: SubmitAnswerRequest = serde_json::from_value(json!({ "kind": "yes" })).unwrap();
@ -8096,6 +8260,7 @@ fn answer_from_typed_no_request_maps_to_no_answer() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
};
let req: SubmitAnswerRequest = serde_json::from_value(json!({ "kind": "no" })).unwrap();
@ -8120,6 +8285,7 @@ fn answer_from_typed_selected_request_validates_and_attaches_option() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
};
let req: SubmitAnswerRequest =
serde_json::from_value(json!({ "kind": "selected", "option_key": "approve" })).unwrap();
@ -8160,6 +8326,7 @@ fn answer_from_typed_multi_selected_request_validates_option_keys() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
};
let req: SubmitAnswerRequest = serde_json::from_value(json!({
"kind": "multi_selected",
@ -9577,6 +9744,11 @@ async fn get_run_state_exposes_pending_interviews() {
"allow_freeform": false,
"context_display": null,
"timeout_seconds": null,
"review_target": {
"label": "Quarry review exercise",
"url": "https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
"kind": "document"
},
}),
Some("gate"),
)
@ -9656,6 +9828,11 @@ async fn cache_backed_run_endpoints_reflect_events_appended_after_warmup() {
"allow_freeform": false,
"context_display": null,
"timeout_seconds": null,
"review_target": {
"label": "Quarry review exercise",
"url": "https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
"kind": "document"
},
}),
Some("review"),
)
@ -9713,6 +9890,10 @@ async fn cache_backed_run_endpoints_reflect_events_appended_after_warmup() {
state_body["pending_interviews"]["q-cache"]["question"]["text"].as_str(),
Some("Approve cached deploy?")
);
assert_eq!(
state_body["pending_interviews"]["q-cache"]["question"]["review_target"]["url"].as_str(),
Some("https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef")
);
let stages = app
.clone()
@ -9744,6 +9925,10 @@ async fn cache_backed_run_endpoints_reflect_events_appended_after_warmup() {
questions["data"][0]["text"].as_str(),
Some("Approve cached deploy?")
);
assert_eq!(
questions["data"][0]["review_target"]["label"].as_str(),
Some("Quarry review exercise")
);
let settings = app
.clone()
@ -16489,6 +16674,7 @@ async fn list_runs_includes_live_metadata_from_run_state() {
allow_freeform: false,
timeout_seconds: None,
context_display: None,
review_target: None,
},
] {
workflow_event::append_event(&run_store, &run_id, &event)

View file

@ -257,7 +257,7 @@ impl TestAppStateBuilder {
pub fn try_build(mut self) -> anyhow::Result<Arc<AppState>> {
let (store, artifact_store) = self.store_bundle.unwrap_or_else(test_store_bundle);
let vault_path = self.vault_path.unwrap_or_else(test_secret_store_path);
redirect_default_storage_root(&mut self.server_settings, &vault_path);
self.server_settings = redirect_default_storage_root(self.server_settings, &vault_path);
if !self.vault_entries.is_empty() {
let mut vault = Vault::load(vault_path.clone()).expect("test vault should load");
for (name, value) in &self.vault_entries {
@ -683,14 +683,16 @@ pub fn test_secret_store_path() -> PathBuf {
/// machine-dependent, and letting run-creating tests write there.
///
/// Only settings still carrying the production default are redirected; a test
/// that chose its own root keeps it.
fn redirect_default_storage_root(settings: &mut ServerSettings, vault_path: &Path) {
/// that chose its own root keeps it. The redirect goes through
/// [`ServerSettings::with_storage_override`] so the derived local object-store
/// roots move with it instead of pointing back at the real storage tree.
fn redirect_default_storage_root(settings: ServerSettings, vault_path: &Path) -> ServerSettings {
if Path::new(&settings.server.storage.root) != default_storage_dir() {
return;
return settings;
}
let root = vault_path.with_file_name("storage");
std::fs::create_dir_all(&root).expect("test storage root should be creatable");
settings.server.storage.root = root.display().to_string();
settings.with_storage_override(&root)
}
#[must_use]

View file

@ -194,6 +194,12 @@ pub enum GitHubCredentials {
}
impl GitHubCredentials {
/// Whether resolving these credentials mints a new installation token.
#[must_use]
pub fn mints_installation_token(&self) -> bool {
matches!(self, Self::App(_))
}
pub fn from_env(app_id: Option<&str>) -> Result<Option<Self>, String> {
Ok(GitHubAppCredentials::from_env(app_id)?.map(Self::App))
}
@ -372,7 +378,7 @@ pub fn parse_github_owner_repo(url: &str) -> anyhow::Result<(String, String)> {
let path = path.trim_end_matches('/');
let path = path.strip_suffix(".git").unwrap_or(path);
let mut parts = path.splitn(3, '/');
let mut parts = path.split('/');
let owner = parts
.next()
.filter(|s| !s.is_empty())
@ -381,6 +387,9 @@ pub fn parse_github_owner_repo(url: &str) -> anyhow::Result<(String, String)> {
.next()
.filter(|s| !s.is_empty())
.ok_or_else(|| anyhow!("Missing repo in GitHub URL: {display_url}"))?;
if parts.next().is_some() {
bail!("GitHub URL must identify one repository: {display_url}");
}
Ok((owner.to_string(), repo.to_string()))
}
@ -1388,6 +1397,16 @@ mod tests {
assert_eq!(repo, "repo");
}
#[test]
fn parse_rejects_extra_path_components() {
let error = parse_github_owner_repo("https://github.com/owner/repo/issues").unwrap_err();
assert!(
error.to_string().contains("must identify one repository"),
"got: {error}"
);
}
// -----------------------------------------------------------------------
// ssh_url_to_https
// -----------------------------------------------------------------------
@ -2312,6 +2331,24 @@ mod tests {
assert!(!fresh.near_expiry(std::time::Duration::from_mins(15)));
}
#[test]
fn only_app_credentials_mint_installation_tokens() {
let app = GitHubCredentials::App(GitHubAppCredentials {
app_id: "test".to_string(),
private_key_pem: test_rsa_key().to_string(),
slug: None,
});
let pat = GitHubCredentials::Pat("ghp_personal".to_string());
let installation = GitHubCredentials::Installation(InstallationToken {
token: "ghs_installation".to_string(),
expires_at: chrono::Utc::now() + chrono::Duration::minutes(30),
});
assert!(app.mints_installation_token());
assert!(!pat.mints_installation_token());
assert!(!installation.mints_installation_token());
}
#[test]
fn validate_static_github_token_rejects_installation_tokens() {
validate_static_github_token("ghp_personal").unwrap();

View file

@ -1,4 +1,4 @@
use std::collections::HashMap;
use std::collections::{HashMap, HashSet};
use std::time::Duration;
use crate::error::Error;
@ -57,29 +57,44 @@ fn derive_class_from_label(label: &str) -> String {
.collect()
}
fn collect_declared_node_ids(statements: &[Statement], node_ids: &mut HashSet<String>) {
for statement in statements {
match statement {
Statement::Node(node) => {
node_ids.insert(node.id.clone());
}
Statement::Subgraph(subgraph) => {
collect_declared_node_ids(&subgraph.statements, node_ids);
}
_ => {}
}
}
}
struct SemanticState {
graph: Graph,
node_defaults: HashMap<String, AttrValue>,
edge_defaults: HashMap<String, AttrValue>,
graph: Graph,
declared_node_ids: HashSet<String>,
node_defaults: HashMap<String, AttrValue>,
edge_defaults: HashMap<String, AttrValue>,
}
impl SemanticState {
fn new(name: String) -> Self {
fn new(name: String, declared_node_ids: HashSet<String>) -> Self {
Self {
graph: Graph::new(name),
graph: Graph::new(name),
declared_node_ids,
node_defaults: HashMap::new(),
edge_defaults: HashMap::new(),
}
}
fn ensure_node(&mut self, id: &str) {
if !self.graph.nodes.contains_key(id) {
fn ensure_node(&mut self, id: &str) -> &mut Node {
let node_defaults = &self.node_defaults;
self.graph.nodes.entry(id.to_string()).or_insert_with(|| {
let mut node = Node::new(id);
for (k, v) in &self.node_defaults {
node.attrs.insert(k.clone(), v.clone());
}
self.graph.nodes.insert(id.to_string(), node);
}
node.attrs.clone_from(node_defaults);
node
})
}
fn add_class_to_node(node: &mut Node, cls: &str) {
@ -90,12 +105,7 @@ impl SemanticState {
}
fn process_node(&mut self, node_stmt: &NodeStmt, subgraph_class: Option<&str>) {
self.ensure_node(&node_stmt.id);
let node = self
.graph
.nodes
.get_mut(&node_stmt.id)
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
let node = self.ensure_node(&node_stmt.id);
if let Some(attrs) = &node_stmt.attrs {
for (k, v) in attrs {
node.attrs.insert(k.clone(), convert_value(v));
@ -111,11 +121,6 @@ impl SemanticState {
.and_then(AttrValue::as_str)
.map(String::from);
if let Some(class_str) = class_str {
let node = self
.graph
.nodes
.get_mut(&node_stmt.id)
.expect("node was just inserted by ensure_node, so get_mut cannot return None");
for cls in class_str.split(',') {
let cls = cls.trim().to_string();
if !cls.is_empty() && !node.classes.contains(&cls) {
@ -127,12 +132,11 @@ impl SemanticState {
fn process_edge(&mut self, edge_stmt: &EdgeStmt, subgraph_class: Option<&str>) {
for id in &edge_stmt.nodes {
self.ensure_node(id);
if !self.declared_node_ids.contains(id) {
continue;
}
let node = self.ensure_node(id);
if let Some(cls) = subgraph_class {
let node =
self.graph.nodes.get_mut(id).expect(
"node was just inserted by ensure_node, so get_mut cannot return None",
);
Self::add_class_to_node(node, cls);
}
}
@ -251,7 +255,9 @@ impl SemanticState {
///
/// Returns an error if the AST cannot be converted to a valid graph.
pub fn ast_to_graph(dot: &DotGraph) -> Result<Graph, Error> {
let mut state = SemanticState::new(dot.name.clone());
let mut declared_node_ids = HashSet::new();
collect_declared_node_ids(&dot.statements, &mut declared_node_ids);
let mut state = SemanticState::new(dot.name.clone(), declared_node_ids);
let empty = HashMap::new();
state.process_statements(&dot.statements, None, &empty, &empty);
Ok(state.graph)
@ -515,7 +521,7 @@ mod tests {
}
#[test]
fn ast_to_graph_implicit_nodes_from_edges() {
fn ast_to_graph_keeps_undeclared_edge_endpoints_out_of_nodes() {
let dot = DotGraph {
name: "Implicit".into(),
statements: vec![Statement::Edge(EdgeStmt {
@ -524,8 +530,83 @@ mod tests {
})],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.is_empty());
assert_eq!(graph.edges, vec![Edge::new("a", "b")]);
}
#[test]
fn ast_to_graph_includes_only_declared_edge_endpoints() {
let dot = DotGraph {
name: "Declared".into(),
statements: vec![
Statement::Node(NodeStmt {
id: "a".into(),
attrs: None,
}),
Statement::Edge(EdgeStmt {
nodes: vec!["a".into(), "b".into()],
attrs: None,
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("a"));
assert!(!graph.nodes.contains_key("b"));
}
#[test]
fn ast_to_graph_declaration_after_edge_still_counts() {
let dot = DotGraph {
name: "DeclaredLater".into(),
statements: vec![
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("first".into()))]),
Statement::Edge(EdgeStmt {
nodes: vec!["a".into(), "b".into()],
attrs: None,
}),
Statement::NodeDefaults(vec![("model".into(), AstValue::Str("second".into()))]),
Statement::Node(NodeStmt {
id: "b".into(),
attrs: Some(vec![("prompt".into(), AstValue::Str("Do it".into()))]),
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("b"));
assert!(!graph.nodes.contains_key("a"));
assert_eq!(
graph.nodes["b"]
.attrs
.get("model")
.and_then(AttrValue::as_str),
Some("first")
);
}
#[test]
fn ast_to_graph_subgraph_declaration_counts() {
let dot = DotGraph {
name: "SubgraphDeclared".into(),
statements: vec![
Statement::Edge(EdgeStmt {
nodes: vec!["start".into(), "plan".into()],
attrs: None,
}),
Statement::Subgraph(SubgraphStmt {
name: Some("cluster_loop".into()),
statements: vec![Statement::Node(NodeStmt {
id: "plan".into(),
attrs: None,
})],
}),
],
};
let graph = ast_to_graph(&dot).unwrap();
assert!(graph.nodes.contains_key("plan"));
assert!(!graph.nodes.contains_key("start"));
}
}

View file

@ -111,6 +111,18 @@ fn parse_non_tty_freeform_response(prompt_read: PromptRead) -> Answer {
}
}
/// The review target line printed above a question in terminal clients, which
/// cannot render a hyperlink label. The label and resource noun are already in
/// `question.text`, so only the URL is shown. Shared with `fabro-cli`'s attach
/// client.
#[must_use]
pub fn review_target_line(question: &Question) -> Option<String> {
question
.review_target
.as_ref()
.map(|target| format!("Review link: {}", target.url()))
}
/// Ask a multiple-choice question using dialoguer's `Select` widget on a TTY.
fn ask_select_interactive(question: &Question) -> Answer {
let items: Vec<String> = question
@ -237,6 +249,9 @@ impl Interviewer for ConsoleInterviewer {
let rendered = self.styles.render_markdown(context_text);
eprint!("{rendered}");
}
if let Some(line) = review_target_line(&question) {
eprintln!("{line}");
}
let q = question;
let answer = task::spawn_blocking(move || match q.question_type {
QuestionType::MultipleChoice => ask_select_interactive(&q),
@ -251,6 +266,9 @@ impl Interviewer for ConsoleInterviewer {
// Non-TTY fallback: line-based stdin reading
let s = self.styles;
if let Some(line) = review_target_line(&question) {
eprintln!("{line}");
}
eprintln!("{} {}", s.bold_cyan.apply_to("?"), question.text);
let answer = match question.question_type {
@ -314,6 +332,33 @@ mod tests {
assert_eq!(answer.value, AnswerValue::Selected("A".to_string()));
}
#[test]
fn review_target_line_shows_only_the_url() {
let target = fabro_types::ReviewTarget::new(
"Quarry review exercise",
"https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
fabro_types::ReviewTargetKind::Document,
)
.unwrap();
let mut question = Question::new(target.question_text(), QuestionType::MultipleChoice);
question.review_target = Some(target);
assert_eq!(
review_target_line(&question).as_deref(),
Some(
"Review link: \
https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef"
)
);
}
#[test]
fn review_target_line_is_absent_without_a_target() {
let question = Question::new("Approve?", QuestionType::YesNo);
assert_eq!(review_target_line(&question), None);
}
#[test]
fn find_matching_option_by_key_case_insensitive() {
let options = vec![InterviewOption {

View file

@ -10,7 +10,7 @@ mod replay;
use std::collections::HashMap;
use async_trait::async_trait;
use fabro_types::{InterviewOption, Principal, QuestionType, SystemActorKind};
use fabro_types::{InterviewOption, Principal, QuestionType, ReviewTarget, SystemActorKind};
use serde::{Deserialize, Serialize};
use tokio::time;
@ -29,6 +29,8 @@ pub struct Question {
pub metadata: HashMap<String, serde_json::Value>,
#[serde(default)]
pub context_display: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub review_target: Option<ReviewTarget>,
}
impl Question {
@ -44,6 +46,7 @@ impl Question {
stage: String::new(),
metadata: HashMap::new(),
context_display: None,
review_target: None,
}
}
}
@ -219,7 +222,7 @@ pub trait Interviewer: Send + Sync {
// Re-export all implementors at the crate root
pub use auto_approve::AutoApproveInterviewer;
pub use callback::CallbackInterviewer;
pub use console::ConsoleInterviewer;
pub use console::{ConsoleInterviewer, review_target_line};
pub use control::{ControlInterviewer, SubmitError};
pub use control_protocol::{
WORKER_CONTROL_INVALID_CURSOR_REASON, WORKER_CONTROL_PONG_TIMEOUT_REASON,
@ -260,6 +263,7 @@ mod tests {
assert!(q.timeout_seconds.is_none());
assert!(q.stage.is_empty());
assert!(q.metadata.is_empty());
assert!(q.review_target.is_none());
}
#[test]

View file

@ -0,0 +1,400 @@
//! Retry for the first repository clone in a clone-based sandbox.
//!
//! Clone-based providers can mint a GitHub App installation token and clone
//! with it immediately. GitHub can reject that first clone before the token is
//! available to the git endpoint. On a private repository, the rejection can
//! arrive as `Repository not found.` or an authentication failure.
//!
//! Only a token minted during the current clone operation makes those messages
//! safe to retry. Static PATs and pre-minted installation tokens fail fast.
//!
//! Retries reuse the same token on purpose. Replication of a given token only
//! makes progress, so each attempt strictly improves the odds, while re-minting
//! would restart the replication clock.
use std::future::Future;
use std::time::Duration;
use fabro_types::SandboxProviderKind;
use fabro_util::backoff::BackoffPolicy;
use tokio::time;
/// Total clone attempts, including the first.
const MAX_ATTEMPTS: u32 = 3;
/// Why a failed clone attempt is worth repeating.
#[derive(Clone, Copy, Debug, PartialEq, Eq, strum::Display)]
#[strum(serialize_all = "snake_case")]
pub(crate) enum CloneRetryReason {
/// A freshly minted installation token has not reached the GitHub edge
/// cache site serving this clone yet.
TokenReplication,
/// The clone failed on infrastructure, unrelated to credentials.
TransientInfra,
}
/// What a clone failure message tells us about retry safety.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) enum CloneMessageClass {
Retry(CloneRetryReason),
Permanent,
Unknown,
}
impl CloneMessageClass {
pub(crate) fn retry_reason(self) -> Option<CloneRetryReason> {
match self {
Self::Retry(reason) => Some(reason),
Self::Permanent | Self::Unknown => None,
}
}
}
/// Message fragments that mean the clone failed on infrastructure.
///
/// These are safe to retry whether or not the clone was authenticated.
const TRANSIENT_HINTS: &[&str] = &[
"could not resolve host",
"temporary failure in name resolution",
"connection refused",
"connection reset",
"connection timed out",
"timed out",
"network is unreachable",
"no route to host",
"tls handshake",
"early eof",
"rpc failed",
"unexpected disconnect",
"the remote end hung up unexpectedly",
"index-pack failed",
"service unavailable",
"gateway timeout",
"too many requests",
"rate limit",
];
/// Message fragments GitHub uses when a token is not yet visible.
///
/// Only meaningful when the clone carried credentials. The same lag surfaces as
/// 404 or as an auth failure depending on which endpoint answers first.
const TOKEN_REPLICATION_HINTS: &[&str] = &[
"repository not found",
"authentication failed",
"invalid username or password",
"bad credentials",
];
/// Classify a failed clone by its rendered message.
///
/// `token_was_freshly_minted` gates the token-replication reading. A static
/// credential cannot become valid during backoff, so auth failures for it are
/// permanent.
pub(crate) fn classify_message(message: &str, token_was_freshly_minted: bool) -> CloneMessageClass {
let lower = message.to_ascii_lowercase();
if TRANSIENT_HINTS.iter().any(|hint| lower.contains(hint)) {
return CloneMessageClass::Retry(CloneRetryReason::TransientInfra);
}
if TOKEN_REPLICATION_HINTS
.iter()
.any(|hint| lower.contains(hint))
{
return if token_was_freshly_minted {
CloneMessageClass::Retry(CloneRetryReason::TokenReplication)
} else {
CloneMessageClass::Permanent
};
}
let permanent = lower.contains("could not read username")
|| lower.contains("terminal prompts disabled")
|| lower.contains("permission denied")
|| (lower.contains("permission to") && lower.contains("denied"))
|| (lower.contains("destination path") && lower.contains("already exists"))
|| (lower.contains("remote branch") && lower.contains("not found"));
if permanent {
return CloneMessageClass::Permanent;
}
CloneMessageClass::Unknown
}
/// Backoff between clone attempts: 3s, then 9s.
///
/// GitHub's guidance for token replication is to wait a few seconds and retry
/// with the same token. Sub-second delays land inside the same replication
/// window and spend an attempt for nothing.
fn backoff() -> BackoffPolicy {
BackoffPolicy {
initial_delay: Duration::from_secs(3),
factor: 3.0,
max_delay: Duration::from_secs(10),
jitter: false,
}
}
/// Run a clone, repeating it while the failure looks transient.
///
/// `attempt` receives the 1-based attempt number. `classify` decides whether an
/// error is worth repeating; `None` returns it to the caller untouched. When a
/// deadline is present, a retry starts only when its backoff fits before that
/// deadline. The final error is returned as-is.
pub(crate) async fn retry_clone<T, E, Attempt, Fut, Classify>(
provider: SandboxProviderKind,
deadline: Option<time::Instant>,
mut attempt: Attempt,
classify: Classify,
) -> Result<T, E>
where
Attempt: FnMut(u32) -> Fut,
Fut: Future<Output = Result<T, E>>,
Classify: Fn(&E) -> Option<CloneRetryReason>,
{
let backoff = backoff();
for attempt_number in 1..MAX_ATTEMPTS {
match attempt(attempt_number).await {
Ok(value) => return Ok(value),
Err(err) => {
let Some(reason) = classify(&err) else {
return Err(err);
};
let delay = backoff.delay_for_attempt(attempt_number);
if deadline.is_some_and(|deadline| {
delay >= deadline.saturating_duration_since(time::Instant::now())
}) {
return Err(err);
}
// The failure text can carry git stderr, so log the category
// rather than the message. The caller still reports the full
// error if the attempts run out.
tracing::warn!(
provider = %provider,
attempt = attempt_number,
max_attempts = MAX_ATTEMPTS,
reason = %reason,
delay_ms = u64::try_from(delay.as_millis()).unwrap_or(u64::MAX),
"Git clone failed, retrying"
);
time::sleep(delay).await;
}
}
}
attempt(MAX_ATTEMPTS).await
}
#[cfg(test)]
mod tests {
use std::sync::Mutex;
use super::*;
/// Records the attempt numbers a closure was called with.
#[derive(Default)]
struct Attempts(Mutex<Vec<u32>>);
impl Attempts {
fn record(&self, attempt: u32) {
self.0.lock().expect("attempt log mutex").push(attempt);
}
fn recorded(&self) -> Vec<u32> {
self.0.lock().expect("attempt log mutex").clone()
}
}
/// A classifier that treats every failure as worth repeating.
const ALWAYS_RETRY: fn(&String) -> Option<CloneRetryReason> =
|_| Some(CloneRetryReason::TokenReplication);
#[test]
fn private_repo_not_found_after_a_successful_mint_is_a_replication_lag() {
assert_eq!(
classify_message("repository not found: Repository not found.", true),
CloneMessageClass::Retry(CloneRetryReason::TokenReplication)
);
}
#[test]
fn not_found_without_a_fresh_token_is_permanent() {
assert_eq!(
classify_message("repository not found: Repository not found.", false),
CloneMessageClass::Permanent
);
}
#[test]
fn auth_failure_with_a_fresh_token_is_a_replication_lag() {
assert_eq!(
classify_message(
"fatal: Authentication failed for 'https://github.com/owner/repo'",
true
),
CloneMessageClass::Retry(CloneRetryReason::TokenReplication)
);
assert_eq!(
classify_message(
"fatal: Authentication failed for 'https://github.com/owner/repo'",
false
),
CloneMessageClass::Permanent
);
}
#[test]
fn infra_failures_retry_without_credentials() {
for message in [
"fatal: unable to access: Could not resolve host: github.com",
"error: RPC failed; curl 56 recv failure",
"fatal: early EOF",
"Operation timed out",
] {
assert_eq!(
classify_message(message, false),
CloneMessageClass::Retry(CloneRetryReason::TransientInfra),
"expected {message:?} to be transient"
);
}
}
#[test]
fn genuine_failures_are_not_retried() {
for message in [
"fatal: could not read Username for 'https://github.com'",
"remote: Permission to owner/repo.git denied",
"fatal: destination path 'repo' already exists",
] {
assert_eq!(
classify_message(message, true),
CloneMessageClass::Permanent,
"expected {message:?} to fail fast"
);
}
}
#[test]
fn unrecognized_failures_remain_unknown() {
assert_eq!(
classify_message("git clone stopped for an unexpected reason", true),
CloneMessageClass::Unknown
);
}
#[test]
fn backoff_waits_seconds_not_milliseconds() {
let backoff = backoff();
assert_eq!(backoff.delay_for_attempt(1), Duration::from_secs(3));
assert_eq!(backoff.delay_for_attempt(2), Duration::from_secs(9));
}
#[tokio::test(start_paused = true)]
async fn first_success_runs_one_attempt() {
let attempts = Attempts::default();
let result = retry_clone(
SandboxProviderKind::Docker,
None,
|attempt| {
attempts.record(attempt);
async move { Ok::<_, String>(attempt) }
},
ALWAYS_RETRY,
)
.await;
assert_eq!(result, Ok(1));
assert_eq!(attempts.recorded(), vec![1]);
}
#[tokio::test(start_paused = true)]
async fn retries_until_a_later_attempt_succeeds() {
let attempts = Attempts::default();
let result = retry_clone(
SandboxProviderKind::Docker,
None,
|attempt| {
attempts.record(attempt);
async move {
if attempt < 3 {
Err("Repository not found.".to_string())
} else {
Ok(attempt)
}
}
},
ALWAYS_RETRY,
)
.await;
assert_eq!(result, Ok(3));
assert_eq!(attempts.recorded(), vec![1, 2, 3]);
}
#[tokio::test(start_paused = true)]
async fn exhausted_attempts_return_the_final_error() {
let attempts = Attempts::default();
let result = retry_clone(
SandboxProviderKind::Docker,
None,
|attempt| {
attempts.record(attempt);
async move { Err::<(), _>(format!("Repository not found. (attempt {attempt})")) }
},
ALWAYS_RETRY,
)
.await;
assert_eq!(
result,
Err("Repository not found. (attempt 3)".to_string()),
"the caller should see the last failure, not the first"
);
assert_eq!(attempts.recorded(), vec![1, 2, 3]);
}
#[tokio::test(start_paused = true)]
async fn unretryable_failure_stops_immediately() {
let attempts = Attempts::default();
let result = retry_clone(
SandboxProviderKind::Docker,
None,
|attempt| {
attempts.record(attempt);
async move { Err::<(), _>("permission denied".to_string()) }
},
|_: &String| None,
)
.await;
assert_eq!(result, Err("permission denied".to_string()));
assert_eq!(
attempts.recorded(),
vec![1],
"a deterministic failure should not wait out the backoff"
);
}
#[tokio::test(start_paused = true)]
async fn deadline_stops_retry_when_backoff_does_not_fit() {
let attempts = Attempts::default();
let deadline = time::Instant::now() + Duration::from_secs(2);
let result = retry_clone(
SandboxProviderKind::Docker,
Some(deadline),
|attempt| {
attempts.record(attempt);
async move { Err::<(), _>("temporary failure".to_string()) }
},
ALWAYS_RETRY,
)
.await;
assert_eq!(result, Err("temporary failure".to_string()));
assert_eq!(attempts.recorded(), vec![1]);
assert_eq!(time::Instant::now() + Duration::from_secs(2), deadline);
}
}

View file

@ -32,6 +32,8 @@ pub(crate) fn github_repo_layout(
"Clone-based sandboxes currently support GitHub repository origins only: {err}"
))
})?;
validate_path_component("owner", &owner)?;
validate_path_component("repository", &repo)?;
let workspace_root = trim_root(workspace_root);
let repos_root = trim_root(repos_root);
let repos_owner_path = sandbox::join_sandbox_path(repos_root, &owner);
@ -48,6 +50,27 @@ pub(crate) fn github_repo_layout(
})
}
fn validate_path_component(label: &str, component: &str) -> crate::Result<()> {
let is_safe = !matches!(component, "." | "..")
&& component
.bytes()
.all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.'));
if !is_safe {
return Err(crate::Error::message(format!(
"GitHub {label} is not a safe repository path component"
)));
}
Ok(())
}
pub(crate) fn repo_symlink_command(layout: &GitHubRepoLayout) -> String {
format!(
"ln -s {} {}",
sandbox::shell_quote(&layout.primary_repo_path),
sandbox::shell_quote(&layout.primary_repo_link),
)
}
fn trim_root(root: &str) -> &str {
let trimmed = root.trim_end_matches('/');
if trimmed.is_empty() { "/" } else { trimmed }
@ -204,6 +227,37 @@ mod tests {
assert_eq!(layout.execution_directory, "/workspace/fabro");
}
#[test]
fn github_layout_rejects_path_traversal_components() {
for origin in [
"https://github.com/../widgets",
"https://github.com/acme/..",
"https://github.com/%2e%2e/widgets",
] {
let error = github_repo_layout(origin, "/workspace", "/repos")
.expect_err("unsafe path component should fail");
assert!(
error.to_string().contains("safe repository path component"),
"got {error} for {origin}"
);
}
}
#[test]
fn repo_symlink_command_quotes_both_paths() {
let layout = github_repo_layout(
"https://github.com/fabro-sh/fabro",
"/work space",
"/repo root",
)
.unwrap();
assert_eq!(
repo_symlink_command(&layout),
"ln -s '/repo root/fabro-sh/fabro' '/work space/fabro'"
);
}
#[test]
fn record_origin_strips_credentials() {
assert_eq!(

View file

@ -9,13 +9,14 @@ use anyhow::Context as _;
use async_trait::async_trait;
use daytona_api_client::apis::api_keys_api;
use daytona_api_client::apis::configuration::Configuration;
use daytona_api_client::models::SandboxState;
use daytona_api_client::models::api_key_list::Permissions;
use daytona_sdk::api_types::SignedPortPreviewUrl;
use daytona_sdk::toolbox_types::Command as SessionCommandResult;
use daytona_sdk::{DaytonaError, SessionCommandLogsResult};
use fabro_github::GitHubCredentials;
use fabro_static::EnvVars;
use fabro_types::{CommandOutputStream, CommandTermination, RunId};
use fabro_types::{CommandOutputStream, CommandTermination, RunId, SandboxProviderKind};
use fabro_util::time::elapsed_ms;
use rand::Rng;
use tokio::runtime::Handle;
@ -24,6 +25,7 @@ use tokio::task::JoinHandle;
use tokio::{fs, time};
use tokio_util::sync::CancellationToken;
use crate::clone_retry::{self, CloneRetryReason};
use crate::clone_source::{self, CloneDecision, EmptyWorkspaceReason};
use crate::redact::redact_auth_url;
use crate::sandbox::{
@ -61,6 +63,7 @@ pub(crate) const DAYTONA_DASHBOARD_SANDBOXES_URL: &str =
"https://app.daytona.io/dashboard/sandboxes";
const FABRO_SANDBOX_USER_AGENT: &str = concat!("fabro-sandbox/", env!("CARGO_PKG_VERSION"));
const DAYTONA_PROBE_TIMEOUT: Duration = Duration::from_secs(20);
const DAYTONA_START_TIMEOUT: Duration = Duration::from_mins(1);
/// Upper bound on explicit and Drop-triggered Daytona session deletion so a
/// stalled REST call cannot block cancellation/timeout paths indefinitely.
const DAYTONA_SESSION_CLOSE_TIMEOUT: Duration = Duration::from_secs(10);
@ -1048,6 +1051,10 @@ impl Sandbox for DaytonaSandbox {
let layout =
clone_source::github_repo_layout(&origin_url, WORKING_DIRECTORY, REPOS_ROOT)
.map_err(|err| self.fail_init(init_start, err))?;
let token_was_freshly_minted = self
.github_app
.as_ref()
.is_some_and(GitHubCredentials::mints_installation_token);
self.emit(SandboxEvent::GitCloneStarted {
url: origin_url.clone(),
branch: branch.clone(),
@ -1055,40 +1062,26 @@ impl Sandbox for DaytonaSandbox {
let clone_start = Instant::now();
let (username, password) = match &self.github_app {
Some(creds) => {
let (owner, repo) = fabro_github::parse_github_owner_repo(&origin_url)
.map_err(|e| {
let err = crate::Error::message(format!(
"Failed to parse GitHub URL for clone: {e}"
));
self.emit(SandboxEvent::GitCloneFailed {
url: origin_url.clone(),
error: err.to_string(),
causes: err.causes(),
});
err
})?;
fabro_github::resolve_clone_credentials(
&fabro_github::GitHubContext::new(
creds,
&fabro_github::github_api_base_url(),
),
&owner,
&repo,
)
.await
.map_err(|e| {
let err = crate::Error::message(format!(
"Failed to get GitHub App credentials for clone: {e}"
));
self.emit(SandboxEvent::GitCloneFailed {
url: origin_url.clone(),
error: err.to_string(),
causes: err.causes(),
});
self.fail_init(init_start, err)
})?
}
Some(creds) => fabro_github::resolve_clone_credentials(
&fabro_github::GitHubContext::new(
creds,
&fabro_github::github_api_base_url(),
),
&layout.owner,
&layout.repo,
)
.await
.map_err(|e| {
let err = crate::Error::message(format!(
"Failed to get GitHub App credentials for clone: {e}"
));
self.emit(SandboxEvent::GitCloneFailed {
url: origin_url.clone(),
error: err.to_string(),
causes: err.causes(),
});
self.fail_init(init_start, err)
})?,
None => (None, None),
};
@ -1153,19 +1146,24 @@ impl Sandbox for DaytonaSandbox {
self.fail_init(init_start, err)
})?;
let clone_token = password.clone();
let clone_result = git_svc
.clone(
&origin_url,
&layout.primary_repo_path,
daytona_sdk::GitCloneOptions {
branch,
username,
password,
let clone_result = clone_retry::retry_clone(
SandboxProviderKind::Daytona,
None,
|_attempt| {
let git_svc = &git_svc;
let origin = origin_url.as_str();
let target = layout.primary_repo_path.as_str();
let options = daytona_sdk::GitCloneOptions {
branch: branch.clone(),
username: username.clone(),
password: password.clone(),
..Default::default()
},
)
.await;
};
async move { git_svc.clone(origin, target, options).await }
},
|err: &DaytonaError| classify_clone_failure(err, token_was_freshly_minted),
)
.await;
match clone_result {
Ok(()) => {
@ -1179,7 +1177,7 @@ impl Sandbox for DaytonaSandbox {
});
self.fail_init(init_start, err)
})?;
let symlink_cmd = daytona_symlink_command(&layout);
let symlink_cmd = clone_source::repo_symlink_command(&layout);
let symlink_result = process_svc
.execute_command(
&wrap_bash_command(&symlink_cmd),
@ -1231,8 +1229,8 @@ impl Sandbox for DaytonaSandbox {
let _ = self.origin_url.set(origin_url.clone());
self.set_working_directory(layout.execution_directory.clone())
.map_err(|err| self.fail_init(init_start, err))?;
if let Some(token) = clone_token {
match fabro_github::embed_token_in_url(&origin_url, &token) {
if let Some(token) = password.as_deref() {
match fabro_github::embed_token_in_url(&origin_url, token) {
Ok(auth_url) => {
let cmd = format!(
"git -c maintenance.auto=0 remote set-url origin {}",
@ -1367,6 +1365,25 @@ impl Sandbox for DaytonaSandbox {
Ok(())
}
async fn activate(&self) -> crate::Result<()> {
let sandbox = self.sandbox()?;
let current = self.client.get(&sandbox.name).await.map_err(|e| {
crate::Error::context("Failed to inspect Daytona sandbox before activation", e)
})?;
if current.state == Some(SandboxState::Started) {
return Ok(());
}
if current.state == Some(SandboxState::Starting) {
return current
.wait_for_start(Some(DAYTONA_START_TIMEOUT))
.await
.map_err(|e| {
crate::Error::context("Failed to wait for Daytona sandbox activation", e)
});
}
self.start().await
}
async fn stop(&self) -> crate::Result<()> {
self.emit(SandboxEvent::StopStarted {
provider: "daytona".into(),
@ -1515,7 +1532,7 @@ impl Sandbox for DaytonaSandbox {
// Only a GitHub App installation token can be re-minted; a static PAT or
// a pre-minted Installation token is fixed, so re-embedding it changes
// nothing. Short-circuit to Skipped before the resolve + set-url exec.
if !matches!(creds, GitHubCredentials::App(_)) {
if !creds.mints_installation_token() {
return Ok(RefreshOutcome::Skipped);
}
@ -2469,12 +2486,36 @@ fn daytona_bash_session_probe_outcome(execution: crate::Result<ExecResult>) -> c
))
}
fn daytona_symlink_command(layout: &clone_source::GitHubRepoLayout) -> String {
format!(
"ln -s {} {}",
shell_quote(&layout.primary_repo_path),
shell_quote(&layout.primary_repo_link),
)
/// Classify a failed Daytona clone for retry.
///
/// The GitHub 404 does not arrive as an HTTP 404 on the Daytona call. git runs
/// inside the sandbox, so its stderr comes back through the toolbox as the
/// error message — the credential race has to be matched on text. Daytona's own
/// transport failures are visible structurally.
fn classify_clone_failure(
err: &DaytonaError,
token_was_freshly_minted: bool,
) -> Option<CloneRetryReason> {
// A Daytona request timeout does not prove that the remote clone stopped.
// Retrying could overlap the still-running first request.
if matches!(err, DaytonaError::Timeout { .. }) {
return None;
}
match clone_retry::classify_message(err.message(), token_was_freshly_minted) {
clone_retry::CloneMessageClass::Retry(reason) => Some(reason),
clone_retry::CloneMessageClass::Permanent => None,
clone_retry::CloneMessageClass::Unknown => match err {
DaytonaError::RateLimit { .. } => Some(CloneRetryReason::TransientInfra),
DaytonaError::Api { status_code, .. } if (500..600).contains(status_code) => {
Some(CloneRetryReason::TransientInfra)
}
DaytonaError::Timeout { .. }
| DaytonaError::Api { .. }
| DaytonaError::NotFound { .. }
| DaytonaError::General(_) => None,
},
}
}
/// Wrap Bash source in the canonical non-login Bash transport.
@ -2522,10 +2563,12 @@ fn build_bash_session_command(
#[cfg(test)]
mod tests {
use std::sync::atomic::AtomicU32;
use daytona_api_client::models::api_key_list::Permissions;
use fabro_util::error::collect_chain;
use httpmock::Method::GET;
use httpmock::MockServer;
use httpmock::Method::{GET, POST};
use httpmock::{HttpMockResponse, MockServer};
use super::*;
use crate::sandbox::BASH_PROBE_MARKER;
@ -2628,6 +2671,25 @@ mod tests {
})
}
fn sandbox_body(name: &str, state: SandboxState) -> serde_json::Value {
serde_json::json!({
"id": name,
"organizationId": "org-1",
"name": name,
"user": "daytona",
"env": {},
"labels": {},
"public": false,
"networkBlockAll": false,
"target": "us",
"cpu": 2.0,
"gpu": 0.0,
"memory": 4.0,
"disk": 20.0,
"state": state.to_string()
})
}
#[test]
fn daytona_config_defaults() {
let config = DaytonaConfig::default();
@ -2783,6 +2845,107 @@ mod tests {
);
}
#[tokio::test]
async fn activate_skips_start_when_daytona_reports_started() {
let server = MockServer::start_async().await;
let get_sandbox = server
.mock_async(|when, then| {
when.method(GET)
.path("/sandbox/test-sandbox")
.header("authorization", "Bearer dtn_test");
then.status(200)
.header("content-type", "application/json")
.json_body(sandbox_body("test-sandbox", SandboxState::Started));
})
.await;
let start_sandbox = server
.mock_async(|when, then| {
when.method(POST)
.path("/sandbox/test-sandbox/start")
.header("authorization", "Bearer dtn_test");
then.status(200)
.header("content-type", "application/json")
.json_body(sandbox_body("test-sandbox", SandboxState::Started));
})
.await;
let sandbox = mock_daytona_sandbox(&server, "dtn_test", DaytonaConfig::default()).await;
let sdk_sandbox = sandbox
.client
.get("test-sandbox")
.await
.expect("test sandbox should load");
sandbox
.sandbox
.set(sdk_sandbox)
.expect("test sandbox should initialize once");
let get_calls_before = get_sandbox.calls_async().await;
sandbox
.activate()
.await
.expect("an active sandbox should require no restart");
assert_eq!(get_sandbox.calls_async().await, get_calls_before + 1);
start_sandbox.assert_calls_async(0).await;
}
#[tokio::test]
async fn activate_waits_for_a_daytona_start_already_in_progress() {
let server = MockServer::start_async().await;
let response_count = Arc::new(AtomicU32::new(0));
let get_sandbox = server
.mock_async({
let response_count = Arc::clone(&response_count);
move |when, then| {
when.method(GET)
.path("/sandbox/test-sandbox")
.header("authorization", "Bearer dtn_test");
then.respond_with(move |_| {
let state = if response_count.fetch_add(1, Ordering::Relaxed) == 1 {
SandboxState::Starting
} else {
SandboxState::Started
};
HttpMockResponse::builder()
.status(200)
.header("content-type", "application/json")
.body(sandbox_body("test-sandbox", state).to_string())
.build()
});
}
})
.await;
let start_sandbox = server
.mock_async(|when, then| {
when.method(POST)
.path("/sandbox/test-sandbox/start")
.header("authorization", "Bearer dtn_test");
then.status(200)
.header("content-type", "application/json")
.json_body(sandbox_body("test-sandbox", SandboxState::Started));
})
.await;
let sandbox = mock_daytona_sandbox(&server, "dtn_test", DaytonaConfig::default()).await;
let sdk_sandbox = sandbox
.client
.get("test-sandbox")
.await
.expect("test sandbox should load");
sandbox
.sandbox
.set(sdk_sandbox)
.expect("test sandbox should initialize once");
let get_calls_before = get_sandbox.calls_async().await;
sandbox
.activate()
.await
.expect("an in-progress start should be awaited");
assert_eq!(get_sandbox.calls_async().await, get_calls_before + 2);
start_sandbox.assert_calls_async(0).await;
}
#[tokio::test]
async fn base_params_merges_managed_daytona_labels() {
let run_id: RunId = "01HY0000000000000000000000".parse().unwrap();
@ -2845,18 +3008,78 @@ mod tests {
}
#[test]
fn daytona_symlink_command_links_workspace_repo_to_repos_checkout() {
let layout = clone_source::github_repo_layout(
"https://github.com/fabro-sh/fabro",
WORKING_DIRECTORY,
REPOS_ROOT,
)
.unwrap();
fn clone_not_found_after_a_successful_mint_is_retried() {
// The exact error from run 01KYM99DF27JRRW4XSYZBP27K7: git's stderr,
// relayed through the toolbox, five seconds after a token was minted.
let err = DaytonaError::general("repository not found: Repository not found.");
assert_eq!(
daytona_symlink_command(&layout),
"ln -s /home/daytona/repos/fabro-sh/fabro /home/daytona/workspace/fabro"
classify_clone_failure(&err, true),
Some(CloneRetryReason::TokenReplication)
);
assert_eq!(
classify_clone_failure(&err, false),
None,
"without credentials there is no token to replicate"
);
}
#[test]
fn clone_transient_transport_failures_are_retried() {
for err in [
DaytonaError::rate_limit("too many requests"),
DaytonaError::api(503, ""),
] {
assert_eq!(
classify_clone_failure(&err, false),
Some(CloneRetryReason::TransientInfra),
"expected {err:?} to be transient"
);
}
}
#[test]
fn clone_timeout_is_not_retried_without_remote_termination() {
let err = DaytonaError::timeout("request timed out");
assert_eq!(classify_clone_failure(&err, true), None);
}
#[test]
fn clone_api_failure_message_takes_precedence_over_status() {
let not_found = DaytonaError::api(500, "repository not found: Repository not found.");
assert_eq!(
classify_clone_failure(&not_found, true),
Some(CloneRetryReason::TokenReplication)
);
assert_eq!(classify_clone_failure(&not_found, false), None);
for message in [
"fatal: destination path 'fabro' already exists",
"remote: Permission to fabro-sh/fabro.git denied",
] {
assert_eq!(
classify_clone_failure(&DaytonaError::api(500, message), true),
None,
"expected {message:?} to take precedence over HTTP 500"
);
}
}
#[test]
fn clone_client_errors_are_not_retried() {
for err in [
DaytonaError::api(400, "bad request"),
DaytonaError::api(403, "forbidden"),
DaytonaError::not_found("Sandbox not found"),
DaytonaError::general("fatal: could not read Username for 'https://github.com'"),
] {
assert_eq!(
classify_clone_failure(&err, true),
None,
"expected {err:?} to fail fast"
);
}
}
#[test]

View file

@ -3,7 +3,7 @@ use std::fmt::Write as _;
use std::io::Cursor;
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
use std::time::{Instant, SystemTime, UNIX_EPOCH};
use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
use async_trait::async_trait;
use bollard::Docker;
@ -15,9 +15,9 @@ use bollard::container::{
use bollard::errors::Error as DockerError;
use bollard::exec::{CreateExecOptions, StartExecOptions, StartExecResults};
use bollard::image::CreateImageOptions;
use bollard::models::HostConfig;
use bollard::models::{ContainerInspectResponse, HostConfig};
use fabro_github::GitHubCredentials;
use fabro_types::{CommandOutputStream, CommandTermination, RunId};
use fabro_types::{CommandOutputStream, CommandTermination, RunId, SandboxProviderKind};
use fabro_util::time::elapsed_ms;
use futures::StreamExt;
use tokio::io::{AsyncWriteExt, duplex};
@ -37,7 +37,7 @@ use crate::{
CommandOutputCallback, DEFAULT_EXEC_OUTPUT_TAIL_BYTES, DirEntry, ExecResult,
ExecStreamingResult, GrepOptions, Sandbox, SandboxEvent, SandboxEventCallback, SandboxFile,
StderrCollector, StdioProcess, StdioProcessHandle, StdioProcessTermination, WalkOptions,
format_lines_numbered, shell_quote,
clone_retry, format_lines_numbered, shell_quote,
};
const DOCKER_BASH_REQUIREMENT: &str = "Docker sandboxes require /bin/bash for every command, with no `sh` fallback; use an \
@ -46,6 +46,7 @@ const DOCKER_BASH_REQUIREMENT: &str = "Docker sandboxes require /bin/bash for ev
pub(crate) const WORKING_DIRECTORY: &str = "/workspace";
pub(crate) const REPOS_ROOT: &str = "/repos";
const GIT_CLONE_DEPTH: usize = 10;
const GIT_CLONE_TIMEOUT: Duration = Duration::from_mins(5);
#[cfg(test)]
const EXEC_STOP_POLL_SLEEP_SECONDS: &str = "0.005";
#[cfg(not(test))]
@ -55,6 +56,11 @@ const EXEC_TERM_GRACE_SECONDS: &str = "0.02";
#[cfg(not(test))]
const EXEC_TERM_GRACE_SECONDS: &str = "0.2";
struct DockerCloneFailure {
error: crate::Error,
retry_reason: Option<clone_retry::CloneRetryReason>,
}
fn env_entry_name(entry: &str) -> &str {
entry.split_once('=').map_or(entry, |(name, _)| name)
}
@ -145,6 +151,13 @@ enum EnsureImageOutcome {
Pulled,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, strum::Display)]
#[strum(serialize_all = "lowercase")]
enum ContainerStartAction {
Start,
Unpause,
}
impl DockerSandbox {
pub fn new(
config: DockerSandboxOptions,
@ -154,7 +167,25 @@ impl DockerSandbox {
clone_branch: Option<String>,
) -> crate::Result<Self> {
let docker = Docker::connect_with_local_defaults().map_err(crate::Error::docker_connect)?;
Ok(Self {
Ok(Self::with_docker_client(
docker,
config,
github_app,
run_id,
clone_origin_url,
clone_branch,
))
}
fn with_docker_client(
docker: Docker,
config: DockerSandboxOptions,
github_app: Option<GitHubCredentials>,
run_id: Option<RunId>,
clone_origin_url: Option<String>,
clone_branch: Option<String>,
) -> Self {
Self {
docker,
config,
github_app,
@ -169,7 +200,7 @@ impl DockerSandbox {
cached_os_version: std::sync::OnceLock::new(),
rg_available: OnceCell::const_new(),
event_callback: None,
})
}
}
pub async fn reconnect(
@ -656,6 +687,32 @@ impl DockerSandbox {
Ok(())
}
/// Preserve a failed `git clone` result while masking the auth URL.
fn clone_failure_error(
&self,
result: ExecResult,
auth_url: Option<&fabro_redact::DisplaySafeUrl>,
) -> crate::Error {
let error = result
.into_exec_error_with_redactor("git clone", |output| redact_auth_url(output, auth_url));
let message = if self.github_app.is_none() {
"Git clone failed. If this is a private repository, configure a GitHub App with \
`fabro install` and install it for your organization."
} else {
"Failed to clone repository into Docker sandbox"
};
crate::Error::context(message, error)
}
fn report_clone_failure(&self, origin_url: &str, err: crate::Error) -> crate::Error {
self.emit(SandboxEvent::GitCloneFailed {
url: origin_url.to_string(),
error: err.to_string(),
causes: err.causes(),
});
err
}
async fn clone_github_repo(
&self,
origin_url: String,
@ -663,12 +720,10 @@ impl DockerSandbox {
) -> crate::Result<()> {
self.verify_git_available().await?;
let layout = clone_source::github_repo_layout(&origin_url, WORKING_DIRECTORY, REPOS_ROOT)?;
self.emit(SandboxEvent::GitCloneStarted {
url: origin_url.clone(),
branch: branch.clone(),
});
let clone_start = Instant::now();
let token_was_freshly_minted = self
.github_app
.as_ref()
.is_some_and(GitHubCredentials::mints_installation_token);
let auth_url = match &self.github_app {
Some(creds) => Some(
@ -689,26 +744,101 @@ impl DockerSandbox {
.as_ref()
.map_or(origin_url.as_str(), |url| url.as_raw_url().as_str());
let command = git_clone_and_link_command(clone_url, branch.as_deref(), &layout);
self.emit(SandboxEvent::GitCloneStarted {
url: origin_url.clone(),
branch: branch.clone(),
});
let clone_start = Instant::now();
let result = self
.docker_exec_shell(&command, 300_000, Some("/"), None, None)
.await?;
if !result.is_success() {
let stderr = redact_auth_url(&result.stderr, auth_url.as_ref());
let err = crate::Error::message(if self.github_app.is_none() {
format!(
"Git clone failed: {stderr}. If this is a private repository, configure a GitHub App with `fabro install` and install it for your organization."
)
} else {
format!("Failed to clone repo into Docker sandbox: {stderr}")
});
self.emit(SandboxEvent::GitCloneFailed {
url: origin_url,
error: err.to_string(),
causes: err.causes(),
});
return Err(err);
let prepare_command = format!(
"mkdir -p {} {}",
shell_quote(WORKING_DIRECTORY),
shell_quote(&layout.repos_owner_path),
);
match self
.docker_exec_shell(&prepare_command, 10_000, Some("/"), None, None)
.await
{
Ok(result) if result.is_success() => {}
Ok(result) => {
let err = result.into_exec_error("prepare Docker repository checkout");
return Err(self.report_clone_failure(&origin_url, err));
}
Err(err) => {
return Err(self.report_clone_failure(&origin_url, err));
}
}
let command = git_clone_command(clone_url, branch.as_deref(), &layout.primary_repo_path);
let clone_deadline = time::Instant::now() + GIT_CLONE_TIMEOUT;
let clone_result = clone_retry::retry_clone(
SandboxProviderKind::Docker,
Some(clone_deadline),
|_attempt| {
let command = command.as_str();
let auth_url = auth_url.as_ref();
async move {
let remaining = clone_deadline.saturating_duration_since(time::Instant::now());
let timeout_ms = u64::try_from(remaining.as_millis()).unwrap_or(u64::MAX);
if timeout_ms == 0 {
return Err(DockerCloneFailure {
error: crate::Error::message(
"Docker git clone deadline expired before retry",
),
retry_reason: None,
});
}
let result = self
.docker_exec_shell_streaming(
command,
Some(timeout_ms),
Some("/"),
None,
None,
None,
)
.await
.map_err(|error| DockerCloneFailure {
error: crate::Error::context(
"Docker git clone transport failed",
error,
),
retry_reason: None,
})?
.result;
if result.is_success() {
return Ok(());
}
let retry_reason =
classify_docker_clone_result(&result, token_was_freshly_minted);
Err(DockerCloneFailure {
error: self.clone_failure_error(result, auth_url),
retry_reason,
})
}
},
|failure: &DockerCloneFailure| failure.retry_reason,
)
.await;
if let Err(failure) = clone_result {
let err = failure.error;
return Err(self.report_clone_failure(&origin_url, err));
}
let symlink_command = clone_source::repo_symlink_command(&layout);
match self
.docker_exec_shell(&symlink_command, 10_000, Some("/"), None, None)
.await
{
Ok(result) if result.is_success() => {}
Ok(result) => {
let err = result.into_exec_error("create Docker workspace repo symlink");
return Err(self.report_clone_failure(&origin_url, err));
}
Err(err) => {
return Err(self.report_clone_failure(&origin_url, err));
}
}
let _ = self.repo_cloned.set(true);
@ -755,24 +885,26 @@ impl DockerSandbox {
verify_managed_labels(container_id, &labels, self.run_id.as_ref())
}
async fn inspect_labels(&self, container_id: &str) -> crate::Result<HashMap<String, String>> {
let inspect = self
.docker
async fn inspect_container(
&self,
container_id: &str,
) -> crate::Result<ContainerInspectResponse> {
self.docker
.inspect_container(container_id, None::<InspectContainerOptions>)
.await
.map_err(|e| {
if docker_not_found(&e) {
crate::Error::message(format!("Docker container '{container_id}' is gone"))
.map_err(|source| {
let message = if docker_not_found(&source) {
format!("Docker container '{container_id}' is gone")
} else {
crate::Error::message(format!(
"Failed to inspect Docker container '{container_id}': {e}"
))
}
})?;
Ok(inspect
.config
.and_then(|config| config.labels)
.unwrap_or_default())
format!("Failed to inspect Docker container '{container_id}'")
};
crate::Error::context(message, source)
})
}
async fn inspect_labels(&self, container_id: &str) -> crate::Result<HashMap<String, String>> {
let inspect = self.inspect_container(container_id).await?;
Ok(container_labels(&inspect))
}
async fn ensure_name_available(&self) -> crate::Result<Option<String>> {
@ -835,6 +967,67 @@ impl DockerSandbox {
.map_err(|e| crate::Error::context("Failed to upload file to container", e))
}
fn begin_start(&self) -> Instant {
self.emit(SandboxEvent::StartStarted {
provider: "docker".into(),
});
Instant::now()
}
async fn set_container_running(
&self,
container_id: &str,
labels: &HashMap<String, String>,
action: ContainerStartAction,
) -> crate::Result<()> {
let result = match action {
ContainerStartAction::Start => {
self.docker
.start_container(container_id, None::<StartContainerOptions<String>>)
.await
}
ContainerStartAction::Unpause => self.docker.unpause_container(container_id).await,
};
if let Err(source) = result {
if !docker_not_modified(&source) {
return Err(crate::Error::context(
format!(
"Failed to {action} Docker container '{container_id}' with labels {labels:?}"
),
source,
));
}
}
Ok(())
}
async fn complete_start(
&self,
started: Instant,
container_id: &str,
labels: &HashMap<String, String>,
action: ContainerStartAction,
) -> crate::Result<()> {
if let Err(error) = self
.set_container_running(container_id, labels, action)
.await
{
return self.start_error(error);
}
if let Err(error) = self.probe_bash(None).await {
return self.start_error(crate::Error::context(
format!("Docker container '{container_id}' health check"),
error,
));
}
self.emit(SandboxEvent::StartCompleted {
provider: "docker".into(),
duration_ms: elapsed_ms(started),
});
Ok(())
}
fn start_error(&self, error: crate::Error) -> crate::Result<()> {
self.emit(SandboxEvent::StartFailed {
provider: "docker".into(),
@ -1164,19 +1357,17 @@ fn git_clone_command(clone_url: &str, branch: Option<&str>, checkout_path: &str)
command
}
fn git_clone_and_link_command(
clone_url: &str,
branch: Option<&str>,
layout: &clone_source::GitHubRepoLayout,
) -> String {
format!(
"mkdir -p {} {} && {} && ln -s {} {}",
shell_quote(WORKING_DIRECTORY),
shell_quote(&layout.repos_owner_path),
git_clone_command(clone_url, branch, &layout.primary_repo_path),
shell_quote(&layout.primary_repo_path),
shell_quote(&layout.primary_repo_link),
)
fn classify_docker_clone_result(
result: &ExecResult,
token_was_freshly_minted: bool,
) -> Option<clone_retry::CloneRetryReason> {
let stderr = clone_retry::classify_message(&result.stderr, token_was_freshly_minted);
match stderr {
clone_retry::CloneMessageClass::Unknown => {
clone_retry::classify_message(&result.stdout, token_was_freshly_minted).retry_reason()
}
class => class.retry_reason(),
}
}
fn host_config(config: &DockerSandboxOptions) -> HostConfig {
@ -1230,6 +1421,24 @@ fn verify_managed_labels(
Ok(())
}
fn container_labels(inspect: &ContainerInspectResponse) -> HashMap<String, String> {
inspect
.config
.as_ref()
.and_then(|config| config.labels.clone())
.unwrap_or_default()
}
fn activation_action(inspect: &ContainerInspectResponse) -> Option<ContainerStartAction> {
let Some(state) = inspect.state.as_ref() else {
return Some(ContainerStartAction::Start);
};
if state.running != Some(true) {
return Some(ContainerStartAction::Start);
}
(state.paused == Some(true)).then_some(ContainerStartAction::Unpause)
}
fn docker_not_found(error: &DockerError) -> bool {
matches!(error, DockerError::DockerResponseServerError {
status_code: 404,
@ -1418,47 +1627,40 @@ impl Sandbox for DockerSandbox {
}
async fn start(&self) -> crate::Result<()> {
self.emit(SandboxEvent::StartStarted {
provider: "docker".into(),
});
let start = Instant::now();
let started = self.begin_start();
let container_id = self.container_id()?.to_string();
let labels = match self.inspect_labels(&container_id).await {
Ok(labels) => labels,
Err(e) => return self.start_error(e),
let inspect = match self.inspect_container(&container_id).await {
Ok(inspect) => inspect,
Err(error) => return self.start_error(error),
};
if let Err(e) = verify_managed_labels(&container_id, &labels, self.run_id.as_ref()) {
return self.start_error(e);
let labels = container_labels(&inspect);
if let Err(error) = verify_managed_labels(&container_id, &labels, self.run_id.as_ref()) {
return self.start_error(error);
}
if let Err(e) = self
.docker
.start_container(&container_id, None::<StartContainerOptions<String>>)
let action = activation_action(&inspect).unwrap_or(ContainerStartAction::Start);
self.complete_start(started, &container_id, &labels, action)
.await
{
if !docker_not_modified(&e) {
return self.start_error(crate::Error::context(
format!(
"Failed to start Docker container '{container_id}' with labels {labels:?}"
),
e,
));
}
async fn activate(&self) -> crate::Result<()> {
let container_id = self.container_id()?.to_string();
let inspect = self.inspect_container(&container_id).await?;
let labels = container_labels(&inspect);
verify_managed_labels(&container_id, &labels, self.run_id.as_ref())?;
let Some(action) = activation_action(&inspect) else {
return Ok(());
};
match action {
ContainerStartAction::Unpause => {
self.set_container_running(&container_id, &labels, action)
.await
}
ContainerStartAction::Start => {
let started = self.begin_start();
self.complete_start(started, &container_id, &labels, action)
.await
}
}
if let Err(e) = self.probe_bash(None).await {
return self.start_error(crate::Error::context(
format!("Docker container '{container_id}' health check"),
e,
));
}
let duration_ms = elapsed_ms(start);
self.emit(SandboxEvent::StartCompleted {
provider: "docker".into(),
duration_ms,
});
Ok(())
}
async fn stop(&self) -> crate::Result<()> {
@ -1980,7 +2182,7 @@ impl Sandbox for DockerSandbox {
// Only a GitHub App installation token can be re-minted; a static PAT or
// a pre-minted Installation token is fixed, so re-embedding it changes
// nothing. Short-circuit to Skipped before the resolve + set-url exec.
if !matches!(creds, GitHubCredentials::App(_)) {
if !creds.mints_installation_token() {
return Ok(RefreshOutcome::Skipped);
}
@ -2023,6 +2225,9 @@ mod tests {
use std::process::Stdio;
use std::time::Duration;
use bollard::API_DEFAULT_VERSION;
use httpmock::Method::{GET, POST};
use httpmock::MockServer;
use tokio::io::AsyncWriteExt as _;
use tokio::process::Command;
@ -2135,6 +2340,54 @@ mod tests {
));
}
#[tokio::test]
async fn activate_unpauses_paused_running_container() {
let server = MockServer::start_async().await;
let inspect = server
.mock_async(|when, then| {
when.method(GET)
.path_suffix("/containers/test-container/json");
then.status(200)
.header("content-type", "application/json")
.json_body(serde_json::json!({
"Config": {
"Labels": managed_labels::for_run(None)
},
"State": {
"Running": true,
"Paused": true
}
}));
})
.await;
let unpause = server
.mock_async(|when, then| {
when.method(POST)
.path_suffix("/containers/test-container/unpause");
then.status(204);
})
.await;
let start = server
.mock_async(|when, then| {
when.method(POST)
.path_suffix("/containers/test-container/start");
then.status(204);
})
.await;
let docker = Docker::connect_with_http(&server.base_url(), 5, API_DEFAULT_VERSION)
.expect("mock Docker client should connect");
let sandbox = test_docker_sandbox(docker, "test-container");
sandbox
.activate()
.await
.expect("a paused running container should be unpaused");
inspect.assert_calls_async(1).await;
unpause.assert_calls_async(1).await;
start.assert_calls_async(0).await;
}
#[test]
fn default_options_are_clone_based() {
let options = DockerSandboxOptions::default();
@ -2157,20 +2410,16 @@ mod tests {
}
#[test]
fn clone_and_link_command_creates_workspace_symlink_to_repos_checkout() {
let layout = clone_source::github_repo_layout(
"https://github.com/fabro-sh/fabro",
"/workspace",
"/repos",
)
.unwrap();
let command =
git_clone_and_link_command("https://github.com/fabro-sh/fabro", Some("main"), &layout);
fn clone_result_uses_stderr_before_stdout() {
let result = ExecResult {
stdout: "Could not resolve host: github.com".to_string(),
stderr: "fatal: destination path 'fabro' already exists".to_string(),
exit_code: Some(128),
termination: CommandTermination::Exited,
duration_ms: 1,
};
assert_eq!(
command,
"mkdir -p /workspace /repos/fabro-sh && git -c maintenance.auto=0 -c gc.auto=0 clone --branch main --single-branch --depth 10 --no-tags -- https://github.com/fabro-sh/fabro /repos/fabro-sh/fabro && ln -s /repos/fabro-sh/fabro /workspace/fabro"
);
assert_eq!(classify_docker_clone_result(&result, true), None);
}
#[test]
@ -2447,4 +2696,20 @@ mod tests {
entry.read_to_string(&mut content).unwrap();
assert_eq!(content, "hello");
}
fn test_docker_sandbox(docker: Docker, container_id: &str) -> DockerSandbox {
let sandbox = DockerSandbox::with_docker_client(
docker,
DockerSandboxOptions::default(),
None,
None,
None,
None,
);
sandbox
.container_id
.set(container_id.to_string())
.expect("test container should initialize once");
sandbox
}
}

View file

@ -143,7 +143,7 @@ pub(crate) fn classify_exec_failure(stderr: &str) -> Option<&'static str> {
} else if lower.contains("could not resolve host") || lower.contains("network is unreachable") {
Some("network failure inside sandbox - check DNS / egress from the run container")
} else if lower.contains("repository not found") {
Some("github 404 - the App installation may not include this repo")
Some("github 404 - repository is unavailable to the current credentials")
} else if lower.contains("no such remote") && lower.contains("origin") {
Some("origin remote missing - push credentials could not be installed")
} else if lower.contains("not a git repository")

View file

@ -9,6 +9,9 @@ pub mod sandbox_spec;
#[cfg(any(feature = "docker", feature = "daytona"))]
mod clone_source;
#[cfg(any(feature = "docker", feature = "daytona", test))]
mod clone_retry;
#[cfg(any(feature = "docker", feature = "daytona", test))]
mod managed_labels;

View file

@ -835,6 +835,13 @@ impl Sandbox for LocalSandbox {
result
}
async fn activate(&self) -> crate::Result<()> {
// Local sandboxes have no provider resource that can stop or pause.
// Resume paths still call `start()` to recreate the directory and
// verify Bash.
Ok(())
}
async fn git_push_ref(&self, refspec: &str) -> crate::Result<()> {
let has_origin = match self
.exec_command("git remote get-url origin", 10_000, None, None, None)

View file

@ -250,6 +250,10 @@ macro_rules! delegate_sandbox {
self.$field.initialize().await
}
async fn activate(&self) -> $crate::Result<()> {
self.$field.activate().await
}
async fn start(&self) -> $crate::Result<()> {
self.$field.start().await
}
@ -1130,6 +1134,15 @@ pub trait Sandbox: Send + Sync {
remote_path: &str,
) -> crate::Result<()>;
async fn initialize(&self) -> crate::Result<()>;
/// Ensure the provider resource is running and not paused before access.
///
/// This access-time operation must be idempotent. Providers that can stop
/// independently should avoid restarting an already-active sandbox. This
/// lightweight check does not require the full health verification done by
/// [`Sandbox::start`], and it does not keep a sandbox active between calls.
async fn activate(&self) -> crate::Result<()> {
self.start().await
}
async fn start(&self) -> crate::Result<()> {
Ok(())
}

View file

@ -64,7 +64,7 @@ pub async fn open_terminal_for_run(
runtime.clone_branch.clone(),
)
.await?;
sandbox.start().await?;
sandbox.activate().await?;
let api_key = resolve_daytona_api_key(daytona_api_key)?;
let organization_id = resolve_daytona_organization_id(daytona_organization_id);
let session = DaytonaTerminalSession::open(
@ -95,7 +95,7 @@ pub async fn open_terminal_for_run(
run_id,
)
.await?;
sandbox.start().await?;
sandbox.activate().await?;
let session = DockerTerminalSession::open(&sandbox, size).await?;
Ok(Box::new(session))
}

View file

@ -1,5 +1,6 @@
use std::collections::HashMap;
use std::sync::Mutex;
use std::sync::atomic::{AtomicBool, Ordering};
use std::time::Duration;
use async_trait::async_trait;
@ -38,6 +39,9 @@ pub struct MockSandbox {
pub captured_working_dirs: Mutex<Vec<Option<String>>>,
/// Captures the `env_vars` argument from `exec_command` calls.
pub captured_env_vars: Mutex<Option<HashMap<String, String>>>,
pub active: AtomicBool,
pub activate_error: Option<String>,
pub activate_calls: Mutex<u32>,
pub start_calls: Mutex<u32>,
pub stop_calls: Mutex<u32>,
pub delete_calls: Mutex<u32>,
@ -51,6 +55,8 @@ pub struct MockSandbox {
/// filtering.
pub walk_files: Vec<SandboxFile>,
pub walk_files_error: Option<String>,
pub walk_files_called: AtomicBool,
pub walked_while_inactive: AtomicBool,
/// Reported by `exec_command_streaming`. Set to `false` to model a
/// provider that cannot separate stdout from stderr.
pub streams_separated: bool,
@ -70,10 +76,25 @@ impl MockSandbox {
*self.start_calls.lock().expect("start_calls lock poisoned")
}
pub fn activate_count(&self) -> u32 {
*self
.activate_calls
.lock()
.expect("activate_calls lock poisoned")
}
pub fn stop_count(&self) -> u32 {
*self.stop_calls.lock().expect("stop_calls lock poisoned")
}
pub fn walk_files_was_called(&self) -> bool {
self.walk_files_called.load(Ordering::Relaxed)
}
pub fn walked_while_inactive(&self) -> bool {
self.walked_while_inactive.load(Ordering::Relaxed)
}
pub fn delete_count(&self) -> u32 {
*self
.delete_calls
@ -99,6 +120,12 @@ impl MockSandbox {
self.walk_files_error = Some(error.into());
self
}
#[must_use]
pub fn with_activate_error(mut self, error: impl Into<String>) -> Self {
self.activate_error = Some(error.into());
self
}
}
impl MockSandbox {
@ -132,6 +159,9 @@ impl Default for MockSandbox {
captured_commands: Mutex::new(Vec::new()),
captured_working_dirs: Mutex::new(Vec::new()),
captured_env_vars: Mutex::new(None),
active: AtomicBool::new(true),
activate_error: None,
activate_calls: Mutex::new(0),
start_calls: Mutex::new(0),
stop_calls: Mutex::new(0),
delete_calls: Mutex::new(0),
@ -141,6 +171,8 @@ impl Default for MockSandbox {
exec_error: None,
walk_files: Vec::new(),
walk_files_error: None,
walk_files_called: AtomicBool::new(false),
walked_while_inactive: AtomicBool::new(false),
streams_separated: true,
}
}
@ -367,6 +399,11 @@ impl Sandbox for MockSandbox {
relative_start: &str,
options: &WalkOptions,
) -> crate::Result<Vec<SandboxFile>> {
self.walk_files_called.store(true, Ordering::Relaxed);
if !self.active.load(Ordering::Relaxed) {
self.walked_while_inactive.store(true, Ordering::Relaxed);
return Err(crate::Error::message("Sandbox is stopped"));
}
if let Some(error) = &self.walk_files_error {
return Err(crate::Error::message(error.clone()));
}
@ -430,6 +467,7 @@ impl Sandbox for MockSandbox {
}
async fn initialize(&self) -> crate::Result<()> {
self.active.store(true, Ordering::Relaxed);
self.emit(SandboxEvent::Initializing {
provider: "mock".into(),
});
@ -444,13 +482,29 @@ impl Sandbox for MockSandbox {
Ok(())
}
async fn activate(&self) -> crate::Result<()> {
*self
.activate_calls
.lock()
.expect("activate_calls lock poisoned") += 1;
if let Some(error) = &self.activate_error {
return Err(crate::Error::context(
"Mock sandbox activation failed",
std::io::Error::other(error.clone()),
));
}
self.start().await
}
async fn start(&self) -> crate::Result<()> {
*self.start_calls.lock().expect("start_calls lock poisoned") += 1;
self.active.store(true, Ordering::Relaxed);
Ok(())
}
async fn stop(&self) -> crate::Result<()> {
*self.stop_calls.lock().expect("stop_calls lock poisoned") += 1;
self.active.store(false, Ordering::Relaxed);
Ok(())
}

View file

@ -99,7 +99,13 @@ fn truncate_to_limit(text: &str, limit: usize, suffix: &str) -> String {
/// final text is bounded by Slack's section-text limit so even pathological
/// inputs cannot produce `invalid_blocks`.
fn header_section(question: &Question, run_web_url: Option<&str>) -> Value {
let mut text = format!("*{}*", escape_slack_controls(&question.text));
let mut text = question.review_target.as_ref().map_or_else(
|| format!("*{}*", escape_slack_controls(&question.text)),
|target| {
let link = slack_link(target.url(), target.label());
format!("*{}*", target.question_text_with_link(&link))
},
);
if !question.stage.is_empty() {
let _ = write!(
text,
@ -420,9 +426,18 @@ fn lifecycle_pull_request_text(pull_request: &RunLifecyclePullRequest<'_>) -> St
text
}
/// Render `label` as a link to `url` in Slack's `<url|label>` syntax, falling
/// back to plain text when the URL cannot be embedded safely.
///
/// `escape_slack_controls` covers Slack's documented escapes (`&`, `<`, `>`)
/// but Slack has no escape for `|`, which separates the URL from the label.
/// A `|` in the label would split the markup and can make Slack reject the
/// whole block, so it is replaced inside link labels only. Labels can be
/// model-authored (a review target label, for example), so this is reachable.
fn slack_link(url: &str, label: &str) -> String {
if is_safe_slack_link_url(url) {
format!("<{}|{}>", url, escape_slack_controls(label))
let label = escape_slack_controls(label).replace('|', "/");
format!("<{url}|{label}>")
} else {
escape_slack_controls(label)
}
@ -634,6 +649,58 @@ mod tests {
assert!(header.contains("<http://127.0.0.1:32276/runs/run-1|Open in Fabro>"));
}
#[test]
fn header_renders_review_target_as_the_question_link() {
let mut q = Question::new(
"Review the Quarry review exercise document, then choose the next action.",
QuestionType::MultipleChoice,
);
q.review_target = Some(
fabro_types::ReviewTarget::new(
"Quarry review exercise",
"https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
fabro_types::ReviewTargetKind::Document,
)
.unwrap(),
);
let blocks = question_to_blocks("run-1", "q-1", &q, None);
let header = serde_json::to_value(&blocks).unwrap()[0]["text"]["text"]
.as_str()
.unwrap()
.to_string();
assert_eq!(
header,
"*Review the \
<https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef|Quarry review \
exercise> document, then choose the next action.*"
);
}
#[test]
fn review_target_label_pipe_cannot_split_the_slack_link() {
let mut q = Question::new("Review", QuestionType::MultipleChoice);
q.review_target = Some(
fabro_types::ReviewTarget::new(
"Draft | v2",
"https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
fabro_types::ReviewTargetKind::Document,
)
.unwrap(),
);
let blocks = question_to_blocks("run-1", "q-1", &q, None);
let header = serde_json::to_value(&blocks).unwrap()[0]["text"]["text"]
.as_str()
.unwrap()
.to_string();
// Exactly one `|`: the one separating the URL from the label.
assert_eq!(header.matches('|').count(), 1);
assert!(header.contains("|Draft / v2>"));
}
#[test]
fn header_omits_link_when_url_missing() {
let q = Question::new("Approve Plan", QuestionType::YesNo);

View file

@ -19,6 +19,7 @@ use fabro_types::{
SandboxProviderKind, StageCompletion, StageHandler, StageId, StageInferenceProjection,
StageModelUsage, StageOutcome, StageProjection, StageState, StartRecord, SubAgentProjection,
SubAgentStatus, TodoListKind, TodoListProjection, TodoProjection, WorkflowRef, first_event_seq,
timing,
};
use fabro_util::error::render_compact_with_causes;
@ -334,6 +335,7 @@ impl RunProjectionReducer for RunProjection {
allow_freeform: props.allow_freeform,
timeout_seconds: props.timeout_seconds,
context_display: props.context_display.clone(),
review_target: props.review_target.clone(),
},
started_at: ts,
});
@ -405,7 +407,7 @@ impl RunProjectionReducer for RunProjection {
};
stage.response = response;
stage.completion = Some(completion);
stage.timing = Some(props.timing);
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@ -428,7 +430,7 @@ impl RunProjectionReducer for RunProjection {
failure_reason,
timestamp: ts,
});
stage.timing = Some(props.timing);
stage.set_authoritative_timing(props.timing);
if let Some(billing) = &props.billing {
stage.usage.replace_with_billed_usage(billing);
stage.model = Some(billing.model().clone());
@ -449,7 +451,7 @@ impl RunProjectionReducer for RunProjection {
context_window.event_seq = Some(event.seq);
stage.context_window = Some(context_window);
}
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentLlmStarted(props) => {
open_inference_bracket(self, stored, props, event.seq, ts);
@ -478,10 +480,10 @@ impl RunProjectionReducer for RunProjection {
inference.first_output_kind = None;
}
EventBody::AgentError(props) => {
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSessionEnded(_) => {
close_inference_brackets_for_session(self, stored);
close_active_brackets_for_session(self, stored, ts);
}
EventBody::AgentSessionActivated(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@ -497,7 +499,7 @@ impl RunProjectionReducer for RunProjection {
return Ok(());
};
stage.agent_control = AgentControlState::WaitingForSteer;
close_inference_bracket(self, stored, props.visit, event.seq);
close_inference_bracket(self, stored, props.visit, event.seq, ts);
}
EventBody::AgentSteeringInjected(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
@ -520,12 +522,17 @@ impl RunProjectionReducer for RunProjection {
};
stage.agent_tools.clone_from(&props.tools);
}
// `AgentAcpStarted` is the start-of-process signal for an external
// ACP agent. `provider_used` is intentionally sourced from the
// subsequent `AgentSessionActivated` event, which carries the
// canonical provider/model. ACP runs without a steering hub never
// emit activation and so legitimately leave `provider_used`
// unset — matching legacy ACP behavior.
EventBody::AgentAcpStarted(props) => {
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
};
stage.open_acp_inference(ts);
// `provider_used` is intentionally sourced from the subsequent
// `AgentSessionActivated` event, which carries the canonical
// provider/model. ACP runs without a steering hub never emit
// activation and legitimately leave it unset.
}
EventBody::CommandStarted(props) => {
let script_invocation = serde_json::to_value(props).map_err(|err| {
Error::InvalidEvent(format!("invalid command.started payload: {err}"))
@ -552,6 +559,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -564,6 +572,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -576,6 +585,7 @@ impl RunProjectionReducer for RunProjection {
let Some(stage) = stage_at_current_visit(self, stored, event.seq) else {
return Ok(());
};
stage.close_acp_inference(props.duration_ms);
apply_agent_terminal(
"agent.acp",
stage,
@ -594,6 +604,11 @@ impl RunProjectionReducer for RunProjection {
// Branches bypass the engine's StageStarted/StageCompleted
// lifecycle. Seed started_at so the branch stage drives a live
// wall-clock timer while it runs (the entry is created Running).
let handler = stored
.node_id
.as_deref()
.and_then(|node_id| self.spec().graph.nodes.get(node_id))
.map(|node| StageHandler::from_handler_type(node.handler_type()));
let is_new = stored
.stage_id
.as_ref()
@ -604,6 +619,9 @@ impl RunProjectionReducer for RunProjection {
if stage.started_at.is_none() {
stage.started_at = Some(ts);
}
if stage.handler.is_none() {
stage.handler = handler;
}
if is_new {
stage.graph_visit = props.graph_visit;
stage
@ -746,6 +764,11 @@ impl RunProjectionReducer for RunProjection {
});
}
EventBody::AgentToolStarted(props) => {
let root_session_id = if stored.parent_session_id.is_none() {
stored.session_id.clone()
} else {
None
};
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
@ -766,6 +789,25 @@ impl RunProjectionReducer for RunProjection {
projection.invoked = true;
}
}
// A subagent's tools run inside the root session's tool call,
// so the root batch already covers them. Timing them again
// would double-count that span.
if let Some(session_id) = root_session_id {
stage.open_tool_call(session_id, props.tool_call_id.clone(), ts);
}
}
EventBody::AgentToolCompleted(props) => {
if stored.parent_session_id.is_some() {
return Ok(());
}
let Some(session_id) = stored.session_id.as_deref() else {
return Ok(());
};
let Some(stage) = stage_at_stored_or_visit(self, stored, props.visit, event.seq)
else {
return Ok(());
};
stage.close_tool_call(session_id, &props.tool_call_id, ts);
}
_ => {}
}
@ -1031,19 +1073,19 @@ fn open_inference_bracket(
});
}
/// Resolve the stage's `inference` slot when it holds a bracket this event is
/// allowed to mutate.
/// Resolve the stage owning a bracket this event is allowed to mutate,
/// borrowing the whole projection so the caller can also fold elapsed time
/// into the stage's live accumulators.
///
/// `None` when the event came from a child session, when the stage has no
/// open bracket, or when the bracket belongs to a different session — the
/// last case matters after failover, which discards the session and builds a
/// new one within a single visit.
fn matching_inference_slot<'a>(
/// Returns `None` for child-session events, for a stage with no open bracket,
/// and for a bracket belonging to a different session (which is what
/// post-failover events look like).
fn matching_inference_stage<'a>(
state: &'a mut RunProjection,
stored: &RunEvent,
visit: u32,
seq: u32,
) -> Option<&'a mut Option<StageInferenceProjection>> {
) -> Option<&'a mut StageProjection> {
if stored.parent_session_id.is_some() {
return None;
}
@ -1053,7 +1095,7 @@ fn matching_inference_slot<'a>(
.inference
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
opened_here.then_some(&mut stage.inference)
opened_here.then_some(stage)
}
/// Resolve the open inference bracket this event is allowed to mutate.
@ -1063,17 +1105,39 @@ fn matching_inference_bracket<'a>(
visit: u32,
seq: u32,
) -> Option<&'a mut StageInferenceProjection> {
matching_inference_slot(state, stored, visit, seq)?.as_mut()
matching_inference_stage(state, stored, visit, seq)?
.inference
.as_mut()
}
/// Close the bracket on a stage-addressed terminal event.
fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit: u32, seq: u32) {
if let Some(slot) = matching_inference_slot(state, stored, visit, seq) {
*slot = None;
}
/// Close the bracket on a stage-addressed terminal event, folding its elapsed
/// time into the stage's live inference accumulator.
fn close_inference_bracket(
state: &mut RunProjection,
stored: &RunEvent,
visit: u32,
seq: u32,
ts: DateTime<Utc>,
) {
let Some(stage) = matching_inference_stage(state, stored, visit, seq) else {
return;
};
close_bracket_on_stage(stage, ts);
}
/// Close every bracket opened by the session that just ended.
/// Take the open bracket and add its span to `live_inference_ms`.
///
/// Retries inside the bracket are deliberately included: the in-process
/// stopwatch counts a retried attempt's elapsed time as inference, and
/// `agent.llm.retry` keeps the bracket open rather than reopening it.
fn close_bracket_on_stage(stage: &mut StageProjection, ts: DateTime<Utc>) {
let Some(inference) = stage.inference.take() else {
return;
};
stage.accumulate_inference_ms(timing::elapsed_ms(inference.started_at, ts));
}
/// Close every active bracket opened by the session that just ended.
///
/// `agent.session.ended` is the only ordering-safe backstop for terminal
/// cancel and wall-clock timeout, which tear the session down through
@ -1088,7 +1152,11 @@ fn close_inference_bracket(state: &mut RunProjection, stored: &RunEvent, visit:
/// opened. Implemented as a normal stage lookup it would find no target and
/// silently no-op, leaving the bracket open forever on exactly the path it
/// exists to cover.
fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunEvent) {
fn close_active_brackets_for_session(
state: &mut RunProjection,
stored: &RunEvent,
ts: DateTime<Utc>,
) {
if stored.parent_session_id.is_some() {
return;
}
@ -1101,8 +1169,9 @@ fn close_inference_brackets_for_session(state: &mut RunProjection, stored: &RunE
.as_ref()
.is_some_and(|inference| inference.session_id == session_id);
if opened_here {
stage.inference = None;
close_bracket_on_stage(stage, ts);
}
stage.close_tool_batch_for_session(session_id, ts);
}
}
@ -1412,23 +1481,26 @@ fn finalize_unfinished_stages_after_run_failed(
StageState::Failed
};
for (_, stage) in state.iter_stages_mut() {
for (_, stage) in state.iter_stages_unordered_mut() {
if stage.state.is_terminal() {
continue;
}
// Close any brackets still open so their spans are not dropped on the
// floor when the live estimate is frozen into `timing` below.
close_bracket_on_stage(stage, timestamp);
stage.close_open_acp_inference(timestamp);
stage.close_open_tool_batch(timestamp);
// Freeze the live estimate before flipping to a terminal state:
// `live_timing` reads `effective_state` and would return wall-only
// once the stage no longer looks in-flight.
let frozen = stage.live_timing(timestamp);
stage.state = terminal_state;
if stage.timing.is_none() {
if let Some(started_at) = stage.started_at {
let wall_time_ms = u64::try_from(
timestamp
.signed_duration_since(started_at)
.num_milliseconds()
.max(0),
)
.expect("non-negative milliseconds fit in u64");
stage.timing = Some(fabro_types::StageTiming::wall_only(wall_time_ms));
}
if stage.timing.is_none() && stage.started_at.is_some() {
stage.set_authoritative_timing(frozen);
} else {
stage.clear_live_timing();
}
}
}
@ -1545,21 +1617,519 @@ mod tests {
};
use fabro_types::settings::run::{DockerfileSource, EnvironmentProvider};
use fabro_types::{
AgentBackend, AgentControlState, AutomationRef, BilledModelUsage, BilledTokenCounts,
BlockedReason, Checkpoint, CheckpointRecord, CommandTermination, EventBody,
FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Outcome,
PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
AgentBackend, AgentControlState, AttrValue, AutomationRef, BilledModelUsage,
BilledTokenCounts, BlockedReason, Checkpoint, CheckpointRecord, CommandTermination,
EventBody, FailureCategory, FailureDetail, FailureReason, Graph, McpServerStatus, Node,
Outcome, PendingReason, PermissionLevel, PullRequestLink, QuestionType, ReasoningEffort,
RunApprovalState, RunBlobId, RunControlAction, RunDiff, RunEvent, RunSize, RunSpec,
RunStatus, Speed, StageContextWindowBreakdownItem, StageContextWindowCategory,
StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
StageContextWindowWarning, StageModelUsage, StageOutcome, StageState, SubAgentStatus,
SuccessReason, WorkflowSettings, first_event_seq, fixtures, test_support,
StageContextWindowWarning, StageHandler, StageModelUsage, StageOutcome, StageState,
StageTiming, SubAgentStatus, SuccessReason, WorkflowSettings, first_event_seq, fixtures,
test_support,
};
use serde_json::json;
use super::{RunProjection, RunProjectionReducer, build_summary};
use crate::{Error, EventEnvelope, StageId};
/// Live accumulation of inference and tool time while a stage is in
/// flight. The finalized breakdown still arrives with the terminal event
/// and replaces these; these exist so a long-running stage is not reported
/// as doing no work.
mod live_active_accumulation {
use fabro_types::run_event::{
AgentLlmFirstOutputProps, AgentLlmRetryProps, AgentLlmStartedProps,
AgentToolCompletedProps, AgentToolStartedProps,
};
use fabro_types::{
LlmOutputKind, LlmRetryPhase, ModelRef, Speed, StageOutcome, StageProjection,
};
use super::*;
fn stage_id() -> StageId {
StageId::new("plan", 1)
}
fn session_event(seq: u32, ts: &str, session_id: &str, body: EventBody) -> EventEnvelope {
let mut event = test_stage_event_at(seq, ts, body, stage_id());
event.event.session_id = Some(session_id.to_string());
event
}
fn agent_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
session_event(seq, ts, "session-1", body)
}
/// An event from a sub-agent session nested under the root session.
fn child_event(seq: u32, ts: &str, body: EventBody) -> EventEnvelope {
let mut event = agent_event(seq, ts, body);
event.event.session_id = Some("session-child".to_string());
event.event.parent_session_id = Some("session-1".to_string());
event
}
fn llm_started() -> EventBody {
EventBody::AgentLlmStarted(AgentLlmStartedProps {
requested_model: ModelRef {
provider: "anthropic".parse().unwrap(),
model_id: "claude-fable-5".into(),
speed: Some(Speed::Fast),
},
visit: 1,
})
}
fn tool_started(tool_call_id: &str) -> EventBody {
EventBody::AgentToolStarted(AgentToolStartedProps {
tool_name: "Bash".to_string(),
tool_call_id: tool_call_id.to_string(),
arguments: json!({}),
visit: 1,
tool_call: None,
turn_id: None,
parent_message_id: None,
})
}
fn tool_completed(tool_call_id: &str) -> EventBody {
EventBody::AgentToolCompleted(AgentToolCompletedProps {
tool_name: "Bash".to_string(),
tool_call_id: tool_call_id.to_string(),
output: json!("ok"),
is_error: false,
visit: 1,
tool_result: None,
turn_id: None,
})
}
fn agent_message() -> EventBody {
EventBody::AgentMessage(live_agent_message_props(live_counts(10, 5)))
}
fn started_state() -> RunProjection {
let mut state = initialized_projection();
state
.apply_event(&test_stage_event_at(
1,
"2026-04-07T12:00:00Z",
EventBody::StageStarted(started_props()),
stage_id(),
))
.unwrap();
state
}
fn stage(state: &RunProjection) -> &StageProjection {
state.stage(&stage_id()).unwrap()
}
#[test]
fn closing_an_inference_bracket_accumulates_its_span() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:06Z",
EventBody::AgentLlmFirstOutput(AgentLlmFirstOutputProps {
kind: LlmOutputKind::Text,
visit: 1,
}),
))
.unwrap();
state
.apply_event(&agent_event(4, "2026-04-07T12:00:12Z", agent_message()))
.unwrap();
// 12:00:05 -> 12:00:12; first_output is a marker, not the close.
assert_eq!(stage(&state).live_inference_ms, 7_000);
assert!(stage(&state).inference.is_none());
}
#[test]
fn concurrent_tool_calls_count_once_not_per_call() {
let mut state = started_state();
for (seq, id) in [(2, "call-a"), (3, "call-b"), (4, "call-c")] {
state
.apply_event(&agent_event(seq, "2026-04-07T12:00:00Z", tool_started(id)))
.unwrap();
}
// All three finish 10s later. Summing per-call spans would report
// 30s; the batch actually occupied 10s of wall time.
for (seq, id) in [(5, "call-a"), (6, "call-b"), (7, "call-c")] {
state
.apply_event(&agent_event(
seq,
"2026-04-07T12:00:10Z",
tool_completed(id),
))
.unwrap();
}
assert_eq!(stage(&state).live_tool_ms, 10_000);
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn a_batch_stays_open_until_its_last_call_reports() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:02Z",
tool_started("call-b"),
))
.unwrap();
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:05Z",
tool_completed("call-a"),
))
.unwrap();
assert_eq!(
stage(&state).live_tool_ms,
0,
"batch must not close while call-b is outstanding"
);
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:09Z",
tool_completed("call-b"),
))
.unwrap();
// Measured from the batch open, not from the last call's start.
assert_eq!(stage(&state).live_tool_ms, 9_000);
}
#[test]
fn successive_batches_accumulate() {
let mut state = started_state();
for (seq, ts, body) in [
(2, "2026-04-07T12:00:00Z", tool_started("call-a")),
(3, "2026-04-07T12:00:04Z", tool_completed("call-a")),
(4, "2026-04-07T12:00:10Z", tool_started("call-b")),
(5, "2026-04-07T12:00:16Z", tool_completed("call-b")),
] {
state.apply_event(&agent_event(seq, ts, body)).unwrap();
}
assert_eq!(stage(&state).live_tool_ms, 10_000);
}
#[test]
fn a_duplicate_completion_does_not_drain_the_batch_early() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:00Z",
tool_started("call-b"),
))
.unwrap();
// call-a reports twice, as a replayed or duplicated log can.
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:03Z",
tool_completed("call-a"),
))
.unwrap();
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:04Z",
tool_completed("call-a"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 0);
assert!(stage(&state).tool_batch.is_some());
}
#[test]
fn a_foreign_session_completion_does_not_mutate_the_open_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("call-a"),
))
.unwrap();
state
.apply_event(&session_event(
3,
"2026-04-07T12:00:05Z",
"session-2",
tool_completed("call-a"),
))
.unwrap();
let batch = stage(&state).tool_batch.as_ref().unwrap();
assert_eq!(batch.session_id, "session-1");
assert!(batch.open_call_ids.contains("call-a"));
assert_eq!(stage(&state).live_tool_ms, 0);
}
#[test]
fn a_replacement_session_starts_a_separate_tool_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("old-call"),
))
.unwrap();
state
.apply_event(&session_event(
3,
"2026-04-07T12:00:05Z",
"session-2",
tool_started("new-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 5_000);
let batch = stage(&state).tool_batch.as_ref().unwrap();
assert_eq!(batch.session_id, "session-2");
assert_eq!(
batch.open_call_ids,
["new-call".to_string()].into_iter().collect()
);
// A delayed completion from the old session cannot close the new
// session's batch even when call ids happen to collide.
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:07Z",
tool_completed("new-call"),
))
.unwrap();
assert!(stage(&state).tool_batch.is_some());
state
.apply_event(&session_event(
5,
"2026-04-07T12:00:09Z",
"session-2",
tool_completed("new-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 9_000);
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn subagent_tool_calls_do_not_double_count_against_the_root_batch() {
let mut state = started_state();
state
.apply_event(&agent_event(
2,
"2026-04-07T12:00:00Z",
tool_started("root-call"),
))
.unwrap();
// The sub-agent's own tools run inside the root call's span.
state
.apply_event(&child_event(
3,
"2026-04-07T12:00:01Z",
tool_started("child-call"),
))
.unwrap();
state
.apply_event(&child_event(
4,
"2026-04-07T12:00:02Z",
tool_completed("child-call"),
))
.unwrap();
state
.apply_event(&agent_event(
5,
"2026-04-07T12:00:08Z",
tool_completed("root-call"),
))
.unwrap();
assert_eq!(stage(&state).live_tool_ms, 8_000);
}
#[test]
fn session_end_accumulates_every_open_active_bracket() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:07Z",
tool_started("call-a"),
))
.unwrap();
let mut ended = test_stage_event_at(
4,
"2026-04-07T12:00:20Z",
EventBody::AgentSessionEnded(AgentSessionEndedProps {}),
stage_id(),
);
ended.event.session_id = Some("session-1".to_string());
state.apply_event(&ended).unwrap();
assert_eq!(stage(&state).live_inference_ms, 15_000);
assert_eq!(stage(&state).live_tool_ms, 13_000);
assert!(stage(&state).inference.is_none());
assert!(stage(&state).tool_batch.is_none());
}
#[test]
fn a_foreign_session_close_leaves_the_bracket_and_accumulator_alone() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:05Z", llm_started()))
.unwrap();
// Post-failover: a new session emits the message, so the old
// bracket is not this event's to close or bill.
let mut foreign =
test_stage_event_at(3, "2026-04-07T12:00:20Z", agent_message(), stage_id());
foreign.event.session_id = Some("session-2".to_string());
state.apply_event(&foreign).unwrap();
assert_eq!(stage(&state).live_inference_ms, 0);
assert!(stage(&state).inference.is_some());
}
#[test]
fn a_retry_keeps_accumulating_within_one_bracket() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(
3,
"2026-04-07T12:00:04Z",
EventBody::AgentLlmRetry(AgentLlmRetryProps {
provider: "anthropic".to_string(),
model: "claude-fable-5".to_string(),
attempt: 0,
delay_secs: 0.0,
error: json!({ "kind": "stream" }),
phase: Some(LlmRetryPhase::Consume),
visit: 1,
}),
))
.unwrap();
state
.apply_event(&agent_event(4, "2026-04-07T12:00:11Z", agent_message()))
.unwrap();
// The whole bracket counts, retry included, matching the
// in-process stopwatch.
assert_eq!(stage(&state).live_inference_ms, 11_000);
assert_eq!(stage(&state).inference, None);
}
#[test]
fn stage_completion_replaces_the_live_estimate_with_finalized_timing() {
let mut state = started_state();
state
.apply_event(&agent_event(2, "2026-04-07T12:00:00Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(3, "2026-04-07T12:00:09Z", agent_message()))
.unwrap();
assert_eq!(stage(&state).live_inference_ms, 9_000);
state
.apply_event(&test_stage_event_at(
4,
"2026-04-07T12:00:10Z",
EventBody::StageCompleted(completed_props(10_000, StageOutcome::Succeeded)),
stage_id(),
))
.unwrap();
let stage = stage(&state);
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:30:00Z")),
stage.timing.unwrap(),
"a terminal stage reports its finalized breakdown, not a live estimate"
);
assert_eq!(stage.live_inference_ms, 0);
assert_eq!(stage.live_tool_ms, 0);
assert!(stage.inference.is_none());
assert!(stage.tool_batch.is_none());
}
#[test]
fn run_failure_freezes_open_work_and_clears_live_bookkeeping() {
let mut state = started_state();
state.status = RunStatus::Running;
state
.apply_event(&agent_event(2, "2026-04-07T12:00:01Z", llm_started()))
.unwrap();
state
.apply_event(&agent_event(3, "2026-04-07T12:00:04Z", agent_message()))
.unwrap();
state
.apply_event(&agent_event(
4,
"2026-04-07T12:00:05Z",
tool_started("call-a"),
))
.unwrap();
let mut failed = test_event(
5,
EventBody::RunFailed(run_failed_props(FailureReason::WorkflowError)),
None,
);
failed.event.ts = test_dt("2026-04-07T12:00:10Z");
state.apply_event(&failed).unwrap();
let stage = stage(&state);
assert_eq!(
stage.timing,
Some(fabro_types::StageTiming::new(10_000, 3_000, 5_000))
);
assert_eq!(stage.state, StageState::Failed);
assert_eq!(stage.live_inference_ms, 0);
assert_eq!(stage.live_tool_ms, 0);
assert!(stage.inference.is_none());
assert!(stage.tool_batch.is_none());
}
}
fn test_event(seq: u32, body: EventBody, node_id: Option<&str>) -> EventEnvelope {
let event = RunEvent {
id: format!("evt-{seq}"),
@ -2365,6 +2935,49 @@ mod tests {
);
}
#[test]
fn parallel_branch_started_uses_the_graph_handler_for_live_timing() {
for (handler_type, handler, expected) in [
(
"prompt",
StageHandler::Prompt,
StageTiming::new(5_000, 5_000, 0),
),
(
"command",
StageHandler::Command,
StageTiming::new(5_000, 0, 5_000),
),
] {
let mut spec = test_run_spec();
let mut node = Node::new("review");
node.attrs.insert(
"type".to_string(),
AttrValue::String(handler_type.to_string()),
);
spec.graph.nodes.insert(node.id.clone(), node);
let mut state = RunProjection::new("Test run".to_string(), spec, Utc::now());
let branch = StageId::new("review", 1);
state
.apply_event(&test_stage_event_at(
3,
"2026-04-07T12:00:00Z",
EventBody::ParallelBranchStarted(ParallelBranchStartedProps {
index: 0,
graph_visit: None,
resumed_from_stage_id: None,
}),
branch.clone(),
))
.unwrap();
let stage = state.stage(&branch).unwrap();
assert_eq!(stage.handler, Some(handler));
assert_eq!(stage.live_timing(test_dt("2026-04-07T12:00:05Z")), expected);
}
}
fn start_stage(state: &mut RunProjection, stage_id: &StageId) {
state
.apply_event(&test_stage_event(
@ -2509,6 +3122,66 @@ mod tests {
assert_eq!(provider_used.model.as_deref(), Some("fake"));
}
#[test]
fn acp_events_accumulate_live_inference_time() {
let mut state = initialized_projection();
let stage_id = StageId::new("code", 1);
state
.apply_event(&test_stage_event_at(
3,
"2026-04-07T12:00:00Z",
EventBody::StageStarted(StageStartedProps {
graph_visit: None,
resumed_from_stage_id: None,
index: 0,
handler_type: "agent".to_string(),
attempt: 1,
max_attempts: 1,
}),
stage_id.clone(),
))
.unwrap();
state
.apply_event(&test_stage_event_at(
4,
"2026-04-07T12:00:05Z",
EventBody::AgentAcpStarted(AgentAcpStartedProps {
visit: 1,
command: "python fake_agent.py".to_string(),
config_name: Some("fake".to_string()),
}),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:00:15Z")),
StageTiming::new(15_000, 10_000, 0)
);
state
.apply_event(&test_stage_event_at(
5,
"2026-04-07T12:00:17Z",
EventBody::AgentAcpCompleted(AgentAcpCompletedProps {
stdout: "done".to_string(),
stderr: String::new(),
stop_reason: "end_turn".to_string(),
duration_ms: 12_000,
}),
stage_id.clone(),
))
.unwrap();
let stage = state.stage(&stage_id).unwrap();
assert_eq!(
stage.live_timing(test_dt("2026-04-07T12:00:20Z")),
StageTiming::new(20_000, 12_000, 0)
);
}
#[test]
fn agent_acp_completed_updates_stage_output_projection() {
let mut state = initialized_projection();
@ -2859,6 +3532,14 @@ mod tests {
allow_freeform: true,
timeout_seconds: Some(30.0),
context_display: Some("Latest draft".to_string()),
review_target: Some(
fabro_types::ReviewTarget::new(
"Quarry review exercise",
"https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef",
fabro_types::ReviewTargetKind::Document,
)
.unwrap(),
),
}),
Some("gate"),
))
@ -2886,6 +3567,14 @@ mod tests {
pending.question.context_display.as_deref(),
Some("Latest draft")
);
assert_eq!(
pending
.question
.review_target
.as_ref()
.map(fabro_types::ReviewTarget::url),
Some("https://quarry.lithos.computer/tmp/0123456789abcdef0123456789abcdef")
);
state
.apply_event(&test_event(

View file

@ -3,7 +3,7 @@ use std::fmt::Write as _;
use std::sync::LazyLock;
use chrono::{DateTime, Utc};
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming};
use fabro_types::{BilledTokenCounts, Run, RunId, RunSize, RunStatusKind, RunTiming, timing};
use sqlx::sqlite::{SqliteConnection, SqliteRow};
use sqlx::{QueryBuilder, Row as _, Sqlite, SqlitePool};
use strum::VariantArray as _;
@ -545,7 +545,7 @@ fn overlay_live_wall_time(run: &mut Run, now: DateTime<Utc>) {
let Some(started_at) = run.timestamps.started_at else {
return;
};
let wall_time_ms = RunTiming::wall_time_ms_since(started_at, now);
let wall_time_ms = timing::elapsed_ms(started_at, now);
run.timing = Some(
run.timing
.unwrap_or_else(|| RunTiming::wall_only(wall_time_ms))

View file

@ -327,6 +327,18 @@ impl Database {
Ok(self.projection_cache.get(run_id).await)
}
pub async fn get_cached_projection(
&self,
run_id: &RunId,
) -> Result<Option<Arc<RunProjection>>> {
self.warm_projection_cache().await?;
Ok(self
.projection_cache
.projection_snapshot(run_id)
.await
.map(|(projection, _)| projection))
}
pub async fn get_cached_summary(
&self,
run_id: &RunId,

View file

@ -235,6 +235,7 @@ fn projection_query_methods_expose_common_state() {
allow_freeform: true,
timeout_seconds: None,
context_display: None,
review_target: None,
},
started_at: Utc::now(),
})]);

View file

@ -1,3 +1,5 @@
use std::collections::HashSet;
use fabro_graphviz::graph::Graph;
use crate::{Diagnostic, LintRule, Severity};
@ -8,6 +10,26 @@ pub(super) fn rule() -> Box<dyn LintRule> {
struct Rule;
impl Rule {
fn diagnostic(&self, node_id: &str, from: &str, to: &str) -> Diagnostic {
Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!(
"Node '{node_id}' is referenced by edge '{from} -> {to}' but has no node \
declaration"
),
node_id: Some(node_id.to_string()),
edge: Some((from.to_string(), to.to_string())),
fix: Some(format!(
"Declare node '{node_id}' or correct the edge endpoint"
)),
..Diagnostic::default()
}
}
}
impl LintRule for Rule {
fn name(&self) -> &'static str {
"edge_target_exists"
@ -15,36 +37,12 @@ impl LintRule for Rule {
fn apply(&self, graph: &Graph) -> Vec<Diagnostic> {
let mut diagnostics = Vec::new();
let mut reported = HashSet::new();
for edge in &graph.edges {
if !graph.nodes.contains_key(&edge.to) {
diagnostics.push(Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!(
"Edge from '{}' targets non-existent node '{}'",
edge.from, edge.to
),
node_id: None,
edge: Some((edge.from.clone(), edge.to.clone())),
fix: Some(format!("Define node '{}' or fix the edge target", edge.to)),
..Diagnostic::default()
});
}
if !graph.nodes.contains_key(&edge.from) {
diagnostics.push(Diagnostic {
rule: self.name().to_string(),
severity: Severity::Error,
message: format!("Edge source '{}' references non-existent node", edge.from),
node_id: None,
edge: Some((edge.from.clone(), edge.to.clone())),
fix: Some(format!(
"Define node '{}' or fix the edge source",
edge.from
)),
..Diagnostic::default()
});
for endpoint in [&edge.to, &edge.from] {
if !graph.nodes.contains_key(endpoint) && reported.insert(endpoint) {
diagnostics.push(self.diagnostic(endpoint, &edge.from, &edge.to));
}
}
}
diagnostics
@ -53,11 +51,112 @@ impl LintRule for Rule {
#[cfg(test)]
mod tests {
use fabro_graphviz::graph::Edge;
use fabro_graphviz::graph::{Edge, Graph};
use fabro_graphviz::parser;
use super::Rule;
use crate::rules::test_support::minimal_graph;
use crate::{LintRule, Severity};
use crate::{Diagnostic, LintRule, Severity};
fn parse(dot: &str) -> Graph {
parser::parse(dot).expect("fixture should parse")
}
fn undeclared_nodes(graph: &Graph) -> Vec<String> {
Rule.apply(graph)
.into_iter()
.map(|d| d.node_id.expect("diagnostic should name a node"))
.collect()
}
#[test]
fn edge_only_node_is_rejected() {
let graph = parse(
r"digraph EdgeOnly {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> misspelled_node
misspelled_node -> exit
}",
);
let diagnostics = Rule.apply(&graph);
assert_eq!(diagnostics.len(), 1, "diagnostics: {diagnostics:?}");
let Diagnostic {
severity,
node_id,
edge,
..
} = &diagnostics[0];
assert_eq!(*severity, Severity::Error);
assert_eq!(node_id.as_deref(), Some("misspelled_node"));
assert_eq!(
edge.clone(),
Some(("start".to_string(), "misspelled_node".to_string()))
);
}
#[test]
fn declaration_after_the_edge_is_accepted() {
let graph = parse(
r#"digraph DeclaredLater {
start -> work
work [prompt="Do the work"]
work -> exit
start [shape=Mdiamond]
exit [shape=Msquare]
}"#,
);
assert!(Rule.apply(&graph).is_empty());
}
#[test]
fn chained_edges_report_every_undeclared_endpoint() {
let graph = parse(
r"digraph Chained {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> first -> second -> exit
}",
);
assert_eq!(undeclared_nodes(&graph), vec!["first", "second"]);
}
#[test]
fn a_node_is_reported_once_no_matter_how_many_edges_use_it() {
let graph = parse(
r"digraph Repeated {
start [shape=Mdiamond]
exit [shape=Msquare]
start -> typo
typo -> exit
typo -> start
}",
);
assert_eq!(undeclared_nodes(&graph), vec!["typo"]);
}
#[test]
fn subgraph_declaration_is_accepted() {
let graph = parse(
r#"digraph Subgraphed {
start [shape=Mdiamond]
exit [shape=Msquare]
subgraph cluster_loop {
label = "Loop A"
plan [prompt="Plan the work"]
}
start -> plan -> exit
}"#,
);
assert!(Rule.apply(&graph).is_empty());
}
#[test]
fn edge_target_exists_rule_missing_target() {

View file

@ -22,6 +22,7 @@ const HANDLER_SPECIFIC_ATTRS: &[(&str, &[&str])] = &[
("output_retries", &["agent", "prompt"]),
("output_schema", &["agent", "prompt", "command"]),
("prompt", &["agent", "prompt", "parallel.fan_in"]),
("review_target", &["human"]),
];
struct Rule;
@ -197,9 +198,28 @@ mod tests {
"prompt".to_string(),
node_with_attr("prompt", "tab", "output_retries", "2"),
);
g.nodes.insert(
"human".to_string(),
node_with_attr("human", "hexagon", "review_target", "true"),
);
assert!(Rule.apply(&g).is_empty());
}
#[test]
fn warns_on_review_target_on_non_human_node() {
let mut g = minimal_graph();
g.nodes.insert(
"work".to_string(),
node_with_attr("work", "box", "review_target", "true"),
);
let diagnostics = Rule.apply(&g);
assert_eq!(diagnostics.len(), 1);
assert!(diagnostics[0].message.contains("'review_target'"));
assert!(diagnostics[0].message.contains("human"));
}
#[test]
fn accepts_prompt_on_shapeless_node_defaulting_to_agent() {
let mut g = minimal_graph();

View file

@ -1,32 +1,7 @@
use std::borrow::Cow;
use std::collections::HashMap;
use fabro_model::Catalog;
use fabro_types::{
BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageProjection, StageTiming,
};
fn stage_usage_with_cost<'a>(
catalog: Option<&Catalog>,
stage: &'a StageProjection,
) -> Cow<'a, BilledTokenCounts> {
let Some(catalog) = catalog else {
return Cow::Borrowed(&stage.usage);
};
let Some(model) = stage.model.as_ref() else {
return Cow::Borrowed(&stage.usage);
};
if stage.usage.total_usd_micros.is_some() {
return Cow::Borrowed(&stage.usage);
}
let Some(total_usd_micros) = catalog.price_tokens(model, &stage.usage.token_counts()) else {
return Cow::Borrowed(&stage.usage);
};
let mut usage = stage.usage.clone();
usage.total_usd_micros = Some(total_usd_micros);
Cow::Owned(usage)
}
use fabro_types::{BilledTokenCounts, ModelRef, RunProjection, RunTiming, StageTiming};
#[derive(Debug, Clone, PartialEq)]
pub struct ProjectionBillingStage {
@ -80,7 +55,7 @@ pub fn billing_rollup_from_projection(
if is_boundary_stage(projection, stage_id.node_id()) {
continue;
}
let usage = stage_usage_with_cost(catalog, stage);
let usage = stage.billed_usage(catalog);
let usage = usage.as_ref();
if stage.completion.is_none() && stage.timing.is_none() && usage.is_zero() {
continue;

View file

@ -12,6 +12,7 @@ pub mod keys {
pub const PREFERRED_LABEL: &str = "preferred_label";
pub const LAST_STAGE: &str = "last_stage";
pub const LAST_RESPONSE: &str = "last_response";
pub const REVIEW_TARGET: &str = "review_target";
// --- graph.* keys ---
pub const GRAPH_GOAL: &str = "graph.goal";
@ -142,6 +143,7 @@ pub mod keys {
assert!(!is_engine_internal_key("outcome"));
assert!(!is_engine_internal_key("last_stage"));
assert!(!is_engine_internal_key("review.result"));
assert!(!is_engine_internal_key(REVIEW_TARGET));
assert!(!is_engine_internal_key("user.name"));
}
}

View file

@ -423,6 +423,7 @@ fn event_body_from_event(event: &Event) -> EventBody {
allow_freeform,
timeout_seconds,
context_display,
review_target,
} => EventBody::InterviewStarted(fabro_types::InterviewStartedProps {
question_id: question_id.clone(),
question: question.clone(),
@ -432,6 +433,7 @@ fn event_body_from_event(event: &Event) -> EventBody {
allow_freeform: *allow_freeform,
timeout_seconds: *timeout_seconds,
context_display: context_display.clone(),
review_target: review_target.clone(),
}),
Event::InterviewCompleted {
actor: _,

Some files were not shown because too many files have changed in this diff Show more