From 67ed7af02685f11b478cc30f51d1dc2ac173dff4 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 08:38:46 -0400 Subject: [PATCH 1/6] Remove the model request status line above the stage toolbar MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The "Model request · waiting on " readout sat directly above the Chat/Thread/Debug toolbar and appeared and disappeared as requests opened and closed, shifting the toolbar underneath it. Drops the StageInferenceIndicator component and everything that existed only to feed it: the inference/runSettled prop threading through RunStages, and StageActivity's watchdogTimedOut field. The watchdog.timeout event now falls through to the same ignore path it always would have, since it was never in STAGE_ACTIVITY_EVENT_TYPES. The run-events invalidations for watchdog.timeout and agent.llm.* stay: they still refresh stage events for the Debug tab and run state for the insights sidebar. Co-Authored-By: Claude Fable 5 --- .../stage-inference-indicator.test.tsx | 107 ------------------ .../components/stage-inference-indicator.tsx | 87 -------------- apps/fabro-web/app/lib/run-events.test.tsx | 2 +- apps/fabro-web/app/routes/run-stages.test.ts | 13 --- apps/fabro-web/app/routes/run-stages.tsx | 33 ------ 5 files changed, 1 insertion(+), 241 deletions(-) delete mode 100644 apps/fabro-web/app/components/stage-inference-indicator.test.tsx delete mode 100644 apps/fabro-web/app/components/stage-inference-indicator.tsx diff --git a/apps/fabro-web/app/components/stage-inference-indicator.test.tsx b/apps/fabro-web/app/components/stage-inference-indicator.test.tsx deleted file mode 100644 index 45da56ae4..000000000 --- a/apps/fabro-web/app/components/stage-inference-indicator.test.tsx +++ /dev/null @@ -1,107 +0,0 @@ -import { afterEach, beforeEach, describe, expect, test } from "bun:test"; -import TestRenderer, { act } from "react-test-renderer"; - -import { LlmOutputKind } from "@qltysh/fabro-api-client"; -import type { StageInferenceProjection } from "@qltysh/fabro-api-client"; - -import { StageInferenceIndicator } from "./stage-inference-indicator"; - -const OPENED_AT = new Date(Date.now() - 12_000).toISOString(); - -function makeInference( - overrides: Partial = {}, -): StageInferenceProjection { - return { - session_id: "ses_root", - started_at: OPENED_AT, - requested_model: { - provider: "anthropic", - model_id: "claude-fable-5", - }, - retries: 0, - ...overrides, - }; -} - -function render( - inference: StageInferenceProjection | null | undefined, - settled = false, -): string { - let renderer!: TestRenderer.ReactTestRenderer; - act(() => { - renderer = TestRenderer.create( - , - ); - }); - const output = JSON.stringify(renderer.toJSON()); - act(() => renderer.unmount()); - return output; -} - -describe("StageInferenceIndicator", () => { - const actGlobal = globalThis as { - IS_REACT_ACT_ENVIRONMENT?: boolean; - }; - const previousActEnvironment = actGlobal.IS_REACT_ACT_ENVIRONMENT; - - beforeEach(() => { - actGlobal.IS_REACT_ACT_ENVIRONMENT = true; - }); - - afterEach(() => { - if (previousActEnvironment === undefined) { - delete actGlobal.IS_REACT_ACT_ENVIRONMENT; - } else { - actGlobal.IS_REACT_ACT_ENVIRONMENT = previousActEnvironment; - } - }); - - test("renders nothing without an open bracket", () => { - expect(render(undefined)).toBe("null"); - expect(render(null)).toBe("null"); - }); - - test("names the requested model while nothing has come back", () => { - const output = render(makeInference()); - expect(output).toContain("Model request"); - expect(output).toContain("waiting on claude-fable-5"); - expect(output).toContain('"aria-live":"polite"'); - expect(output).toContain('"aria-hidden":"true"'); - // No completion estimate exists, so none may be shown. - expect(output).not.toContain("%"); - }); - - test("reports the observed first-output kind", () => { - expect( - render(makeInference({ first_output_kind: LlmOutputKind.REASONING })), - ).toContain("reasoning"); - expect( - render(makeInference({ first_output_kind: LlmOutputKind.TEXT })), - ).toContain("writing"); - expect( - render(makeInference({ first_output_kind: LlmOutputKind.TOOL_CALL })), - ).toContain("calling tools"); - }); - - test("never says thinking for non-reasoning output", () => { - for (const kind of [LlmOutputKind.TEXT, LlmOutputKind.TOOL_CALL]) { - expect(render(makeInference({ first_output_kind: kind }))).not.toContain( - "thinking", - ); - } - }); - - test("counts retries without presenting them as failure", () => { - const output = render(makeInference({ retries: 2 })); - expect(output).toContain("retry 2"); - expect(output).not.toContain("failed"); - }); - - test("goes static once the run can no longer advance the bracket", () => { - const output = render(makeInference(), true); - // An open bracket on a settled run means we never learned how the request - // ended — animating it would claim work that may not be happening. - expect(output).toContain("never completed"); - expect(output).not.toContain("animate-pulse"); - }); -}); diff --git a/apps/fabro-web/app/components/stage-inference-indicator.tsx b/apps/fabro-web/app/components/stage-inference-indicator.tsx deleted file mode 100644 index a329f3fa0..000000000 --- a/apps/fabro-web/app/components/stage-inference-indicator.tsx +++ /dev/null @@ -1,87 +0,0 @@ -import { LlmOutputKind } from "@qltysh/fabro-api-client"; -import type { StageInferenceProjection } from "@qltysh/fabro-api-client"; - -import { Tooltip } from "./ui"; -import { formatAbsoluteTs, formatDurationSecs } from "../lib/format"; -import { elapsedSecsSince, useTickingNow } from "../lib/time"; - -export interface StageInferenceIndicatorProps { - /** Open inference bracket from the stage projection, if there is one. */ - inference: StageInferenceProjection | null | undefined; - /** - * The run can no longer make progress on this bracket: it reached a terminal - * status, or the stall watchdog fired. An open bracket then means *we never - * learned how the request ended*, not *it is still working*, so the readout - * goes static. - */ - settled: boolean; -} - -const ACTIVITY_LABEL: Record = { - [LlmOutputKind.REASONING]: "reasoning", - [LlmOutputKind.TEXT]: "writing", - [LlmOutputKind.TOOL_CALL]: "calling tools", -}; - -/** - * Live readout for an open model request. - * - * Says only what the event log proves. There is no progress bar, percentage, - * or ETA, because no completion estimate exists; the elapsed clock counts - * since the request opened rather than claiming the model is still working; - * and "reasoning" appears only when the provider actually sent reasoning - * output, never as a guess filling a gap in the log. - */ -export function StageInferenceIndicator({ - inference, - settled, -}: StageInferenceIndicatorProps) { - // Ticking is what distinguishes "we are still hearing from this request" - // from "this is a record of one that never closed", so it stops the moment - // the bracket can no longer advance. - const now = useTickingNow(Boolean(inference) && !settled); - - if (!inference) return null; - - if (settled) { - return ( -

- Model request opened {formatAbsoluteTs(inference.started_at)}, never - completed -

- ); - } - - const elapsedSecs = elapsedSecsSince(inference.started_at, now); - const activity = inference.first_output_kind - ? ACTIVITY_LABEL[inference.first_output_kind] - : `waiting on ${inference.requested_model.model_id}`; - - const statusParts = ["Model request", activity]; - // A retry that later succeeds is normal, so this is a count, not a failure. - if (inference.retries > 0) { - statusParts.push(`retry ${inference.retries}`); - } - - return ( -

- - - - -

- ); -} diff --git a/apps/fabro-web/app/lib/run-events.test.tsx b/apps/fabro-web/app/lib/run-events.test.tsx index 11544b14d..e984be74d 100644 --- a/apps/fabro-web/app/lib/run-events.test.tsx +++ b/apps/fabro-web/app/lib/run-events.test.tsx @@ -150,7 +150,7 @@ describe("queryKeysForRunEvent", () => { ]); }); - test("watchdog timeout refreshes the stage events that settle inference", () => { + test("watchdog timeout refreshes the stage events for that stage", () => { expect( queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"), ).toEqual([queryKeys.runs.stageEvents("run-1", "code@1")]); diff --git a/apps/fabro-web/app/routes/run-stages.test.ts b/apps/fabro-web/app/routes/run-stages.test.ts index 2238aaf74..51684a895 100644 --- a/apps/fabro-web/app/routes/run-stages.test.ts +++ b/apps/fabro-web/app/routes/run-stages.test.ts @@ -884,19 +884,6 @@ describe("buildChatItems", () => { }); describe("buildStageActivity pending tools", () => { - test("records watchdog settlement only for the selected stage", () => { - const events: EventEnvelope[] = [ - envelope(1, { - event: "watchdog.timeout", - stage_id: "plan@1", - node_id: "plan", - }), - ]; - - expect(buildStageActivity(events, "plan@1").watchdogTimedOut).toBe(true); - expect(buildStageActivity(events, "code@1").watchdogTimedOut).toBe(false); - }); - test("returns started-but-not-completed calls for the stage", () => { const events: EventEnvelope[] = [ envelope(1, { diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index 48111ce23..ba2432151 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -31,7 +31,6 @@ import type { ThreadDnaSelection, } from "../components/event-debug"; import { StageContext } from "../components/stage-context"; -import { StageInferenceIndicator } from "../components/stage-inference-indicator"; import { StageInsightsSidebar } from "../components/stage-insights-sidebar"; import { StageSidebar } from "../components/stage-sidebar"; import type { Stage } from "../components/stage-sidebar"; @@ -73,7 +72,6 @@ import { useRunStages, useRunState, } from "../lib/queries"; -import { isTerminalRunStatus } from "../lib/run-actions"; import { STAGE_ACTIVITY_EVENT_TYPES, type StageActivityEventType, @@ -86,7 +84,6 @@ import { getNumber, getString, type UnknownRecord } from "../lib/unknown"; import type { EventEnvelope, StageHandler, - StageInferenceProjection, StageModelUsage, } from "@qltysh/fabro-api-client"; @@ -271,7 +268,6 @@ export interface PendingToolCall { interface StageActivity { turns: TurnType[]; pendingTools: PendingToolCall[]; - watchdogTimedOut: boolean; } interface PendingCommand { @@ -287,17 +283,12 @@ export function buildStageActivity( const pendingTools = new Map(); let pendingCommand: PendingCommand | undefined; let sawAssistantMessage = false; - let watchdogTimedOut = false; for (const e of events) { const eventName = e.event; if (activityEventStageId(e) !== stageId) { continue; } - if (eventName === "watchdog.timeout") { - watchdogTimedOut = true; - continue; - } if ( !eventName || !STAGE_ACTIVITY_EVENT_SET.has(eventName) @@ -452,7 +443,6 @@ export function buildStageActivity( return { turns, - watchdogTimedOut, pendingTools: Array.from(pendingTools, ([toolCallId, tool]) => ({ toolCallId, toolName: tool.toolName, @@ -2075,8 +2065,6 @@ function RunStageActivityStage({ selectedStage, stages, runStart, - inference, - runSettled, tab, selectedKinds, selectedDebugCategories, @@ -2090,8 +2078,6 @@ function RunStageActivityStage({ selectedStage: Stage; stages: Stage[]; runStart: string | undefined; - inference: StageInferenceProjection | null | undefined; - runSettled: boolean; tab: EventsTab; selectedKinds: EventKind[]; selectedDebugCategories: DebugCategory[]; @@ -2103,15 +2089,11 @@ function RunStageActivityStage({ }) { const selectedStageId = selectedStage.id; const stageEventsQuery = useRunStageEvents(runId, selectedStageId); - // An open bracket on a run that can no longer advance means we never learned - // how the request ended, not that it is still working. The watchdog stays - // the authority on "actually stuck", so its timeout settles the readout too. const activity = useMemo( () => buildStageActivity(stageEventsQuery.data ?? [], selectedStageId), [stageEventsQuery.data, selectedStageId], ); const { turns } = activity; - const inferenceSettled = runSettled || activity.watchdogTimedOut; const renderer: StageRenderer = selectStageRenderer(selectedStage.handler); const debugEvents = useMemo(() => { return (stageEventsQuery.data ?? []).filter( @@ -2259,11 +2241,6 @@ function RunStageActivityStage({

)} - - ); From 73894c0f767b7e8b23b5e758e99236676b63dcc7 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 08:39:38 -0400 Subject: [PATCH 2/6] Default the stage activity panel to the Chat tab Chat is the more useful first view for agent stages, so open there instead of Thread. Only agent stages offer "chat" in availableTabs; every other renderer already falls back to "primary", so this leaves Logs/Q&A/Decision and the rest unchanged. Co-Authored-By: Claude Fable 5 --- apps/fabro-web/app/routes/run-stages.tsx | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index ba2432151..22b862716 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -178,7 +178,9 @@ type StageActivityAction = | { type: "searchChanged"; search: string }; const initialStageActivityState = (): StageActivityState => ({ - tab: "primary", + // Only agent stages offer "chat"; every other renderer resolves this to + // "primary" through `availableTabs`, so this is the default for both. + tab: "chat", selectedKinds: [...EVENT_KINDS], selectedDebugCategories: [], search: "", From 6b9bf2be8033008bf5ee142b202f921213c4bae6 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 08:42:11 -0400 Subject: [PATCH 3/6] Skip text-free assistant turns in the Chat view A text-free agent message marks the boundary between two batches of tool calls, so it stays in the turn stream to keep those batches as separate "N tool calls" chips. But it rendered an empty prose div, which still took a slot in the gap-4 column and doubled the vertical space between the chips on either side of it. Render nothing for those turns instead. The final assistant turn still renders when it carries a token/duration footer, even with no text, so the completed-stage metrics are unchanged. Co-Authored-By: Claude Fable 5 --- .../app/routes/run-stages-chat.test.tsx | 52 +++++++++++++++++++ apps/fabro-web/app/routes/run-stages.tsx | 12 ++++- 2 files changed, 62 insertions(+), 2 deletions(-) diff --git a/apps/fabro-web/app/routes/run-stages-chat.test.tsx b/apps/fabro-web/app/routes/run-stages-chat.test.tsx index 9ca84db99..be86b89fd 100644 --- a/apps/fabro-web/app/routes/run-stages-chat.test.tsx +++ b/apps/fabro-web/app/routes/run-stages-chat.test.tsx @@ -52,6 +52,58 @@ describe("StageChatView", () => { expect(html).toContain("1m 12s"); }); + test("renders no node for a text-free assistant turn between tool batches", () => { + const html = renderToStaticMarkup( + , + ); + + // The boundary keeps the two batches as separate chips, but contributes + // no element of its own between them. + expect(html.match(/1 tool call/g)).toHaveLength(2); + expect(html).not.toContain('class="prose prose-sm max-w-none"'); + expect(html.match(/class="prose /g)).toHaveLength(1); + expect(html).toContain("Done"); + }); + test("connects a long prompt's expand button to its controlled content", () => { const html = renderToStaticMarkup( ); case "assistant": { + const hasText = turn.content.trim().length > 0; const metric = turnMetric(turn); const isFinal = turnIndex === lastAssistantTurnIndex && !stageActive; + const showFooter = isFinal && Boolean(metric || duration); + // A text-free assistant turn is the boundary between two batches + // of tool calls, kept in the turn stream so those batches stay + // separate chips. It has nothing to show, and an empty node would + // still take a slot in this gap-4 column, doubling the space + // between the chips on either side of it. + if (!hasText && !showFooter) return null; return (
- - {isFinal && (metric || duration) && ( + {hasText && } + {showFooter && (
{metric && {metric}} {duration && {duration}} From d6a66844ff138f0b5e4cfe08bfbd2226dee99614 Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 08:48:33 -0400 Subject: [PATCH 4/6] Keep prompt text inside the Chat bubble MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The prompt bubble is `w-fit max-w-[85%]`, so its width is measured intrinsically and only then clamped. `items-start` left the inner content wrapper intrinsically sized too, so it resolved against the available space from before the clamp — the full column width — and kept that measurement after the bubble shrank. The text laid out at 100% of the column while the background painted at 85%, spilling out the right side. Give the wrapper `w-full` so it fills the bubble's resolved width instead of measuring itself. Short prompts still hug their content: a percentage-width child contributes its content size during intrinsic sizing, so the bubble measures the same and only the final wrap width changes. The expand button keeps hugging its label as a separate flex child. Also break long words in the collapsed preview. That is a separate overflow path: the preview is raw prompt text under `whitespace-pre-wrap`, where an unbreakable path or URL would spill even at the correct width. Co-Authored-By: Claude Fable 5 --- apps/fabro-web/app/routes/run-stages.tsx | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index ad5dfc104..7025541cb 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -1481,11 +1481,18 @@ function ChatUserCard({ content }: { content: string }) { const isLong = content.length > CHAT_PROMPT_PREVIEW_CHARS; return (
-
+ {/* + `w-full` is load-bearing. `items-start` keeps the expand button hugging + its label, but it also leaves this wrapper intrinsically sized, and the + bubble's own `w-fit` width is only clamped to `max-w-[85%]` after that + intrinsic pass. The text then keeps the wider pre-clamp measurement and + spills past the bubble. Filling the resolved width sidesteps it. + */} +
{expanded ? ( ) : ( -

+

{isLong ? `${content.slice(0, CHAT_PROMPT_PREVIEW_CHARS).trimEnd()}…` : content} From 6a96604d0deda172c472893d12517777bc95000a Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Sat, 25 Jul 2026 09:03:15 -0400 Subject: [PATCH 5/6] Show disclosed reasoning in the Thread details panel MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit agent.message already carries a `reasoning` property with the model's own summary and its verbatim trace, and the generated client already types it. The web app just never read it. Read it onto the assistant turn and render it in the details panel, after the message and before the metrics. A trace can run thousands of characters, so leading with one would push the message the user clicked on below the fold. Text over 280 characters collapses to a preview with a "Show all" toggle, matching ChatUserCard's disclosure pattern. Providers disclose one field or the other or both, so a trace with no summary is labeled just "Reasoning" rather than "Reasoning trace" — that is the common Anthropic thinking case, and the bare label reads better when there is nothing to contrast it with. Both fields render as preformatted text: reasoning is raw model output, not authored Markdown, and parsing it would eat the line breaks that are part of what it says. Adding the field to the assistant turn broke six existing toEqual fixtures that assert whole turn objects; they now expect `reasoning: null`. Co-Authored-By: Claude Fable 5 --- .../app/routes/run-stages-details.test.tsx | 72 +++++++++++++++ apps/fabro-web/app/routes/run-stages.test.ts | 43 +++++++++ apps/fabro-web/app/routes/run-stages.tsx | 89 ++++++++++++++++++- 3 files changed, 202 insertions(+), 2 deletions(-) create mode 100644 apps/fabro-web/app/routes/run-stages-details.test.tsx diff --git a/apps/fabro-web/app/routes/run-stages-details.test.tsx b/apps/fabro-web/app/routes/run-stages-details.test.tsx new file mode 100644 index 000000000..d2cd6ad84 --- /dev/null +++ b/apps/fabro-web/app/routes/run-stages-details.test.tsx @@ -0,0 +1,72 @@ +import { describe, expect, test } from "bun:test"; +import { renderToStaticMarkup } from "react-dom/server"; + +import { EventDetails, type TurnReasoning } from "./run-stages"; + +const RUN_START = "2026-04-09T12:00:00Z"; + +function assistantMarkup(reasoning: TurnReasoning | null): string { + return renderToStaticMarkup( + , + ); +} + +describe("EventDetails reasoning", () => { + test("shows nothing when the response disclosed no reasoning", () => { + const html = assistantMarkup(null); + + expect(html).toContain("Refactored the auth module."); + expect(html).not.toContain("Reasoning"); + }); + + test("labels a trace-only response Reasoning, not Reasoning trace", () => { + const html = assistantMarkup({ summary: null, trace: "Considered A." }); + + expect(html).toContain("Reasoning"); + expect(html).not.toContain("Reasoning trace"); + expect(html).toContain("Considered A."); + }); + + test("distinguishes the summary from the verbatim trace when both arrive", () => { + const html = assistantMarkup({ + summary: "Checked the config.", + trace: "Considered A.", + }); + + expect(html).toContain("Reasoning trace"); + expect(html).toContain("Checked the config."); + expect(html).toContain("Considered A."); + }); + + test("renders short reasoning in full, with no disclosure control", () => { + const html = assistantMarkup({ summary: null, trace: "Considered A." }); + + expect(html).not.toContain("Show all"); + expect(html).not.toContain("aria-expanded"); + }); + + test("connects a long trace's expand button to its controlled content", () => { + const trace = "x".repeat(281); + const html = assistantMarkup({ summary: null, trace }); + + const controls = html.match(/aria-controls="([^"]+)"/)?.[1]; + expect(controls).toBeDefined(); + expect(html).toContain(`id="${controls}"`); + expect(html).toContain('aria-expanded="false"'); + expect(html).toContain("Show all (281 characters)"); + // Collapsed, so the preview is truncated rather than the whole trace. + expect(html).not.toContain(trace); + expect(html).toContain(`${"x".repeat(280)}…`); + }); +}); diff --git a/apps/fabro-web/app/routes/run-stages.test.ts b/apps/fabro-web/app/routes/run-stages.test.ts index 51684a895..c77bf73fb 100644 --- a/apps/fabro-web/app/routes/run-stages.test.ts +++ b/apps/fabro-web/app/routes/run-stages.test.ts @@ -114,6 +114,7 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, ]); @@ -131,6 +132,7 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, ]); }); @@ -390,6 +392,7 @@ describe("eventsToActivity", () => { inputTokens: 120, outputTokens: 30, toolCallCount: null, + reasoning: null, }, ]); }); @@ -438,6 +441,7 @@ describe("eventsToActivity", () => { inputTokens: 10, outputTokens: 5, toolCallCount: null, + reasoning: null, }, ]); }); @@ -465,10 +469,48 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 4, toolCallCount: null, + reasoning: null, }, ]); }); + test("reads disclosed reasoning off agent.message", () => { + function reasoningOf(properties: Record) { + const turns = eventsToActivity( + [ + envelope(1, { + event: "agent.message", + stage_id: "plan@1", + node_id: "plan", + properties, + }), + ], + "plan@1", + ); + expect(turns[0].kind).toBe("assistant"); + return turns[0].kind === "assistant" ? turns[0].reasoning : undefined; + } + + expect( + reasoningOf({ + text: "Done.", + reasoning: { summary: "Checked the config", trace: "step one…" }, + }), + ).toEqual({ summary: "Checked the config", trace: "step one…" }); + + // Anthropic thinking arrives as a trace with no summary. + expect(reasoningOf({ text: "Done.", reasoning: { trace: "step one…" } })).toEqual( + { summary: null, trace: "step one…" }, + ); + + expect(reasoningOf({ text: "Done." })).toBe(null); + // A provider that sends the key but nothing usable reads as "none". + expect(reasoningOf({ text: "Done.", reasoning: {} })).toBe(null); + expect( + reasoningOf({ text: "Done.", reasoning: { summary: "", trace: "" } }), + ).toBe(null); + }); + test("formatStageModelUsageLabel includes reasoning effort when present", () => { expect( formatStageModelUsageLabel({ @@ -1216,6 +1258,7 @@ describe("tool-call-only agent responses", () => { inputTokens: 4200, outputTokens: 96, toolCallCount: 2, + reasoning: null, }, ]); }); diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index 7025541cb..9357c17c7 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -80,7 +80,12 @@ import { ACTIVE_STAGE_STATES, mapRunStagesToSidebarStages, } from "../lib/stage-sidebar"; -import { getNumber, getString, type UnknownRecord } from "../lib/unknown"; +import { + getNumber, + getObject, + getString, + type UnknownRecord, +} from "../lib/unknown"; import type { EventEnvelope, StageHandler, @@ -89,6 +94,17 @@ import type { export const handle = { wide: true, fullHeight: true }; +/** + * Readable reasoning a provider disclosed for one model response: `summary` is + * the model's own recap, `trace` its verbatim reasoning. Providers send one, + * the other, or both, and opaque material (signatures, redacted blocks) never + * reaches the wire, so an absent field means nothing was disclosed. + */ +export interface TurnReasoning { + summary: string | null; + trace: string | null; +} + type TurnType = | { kind: "system"; ts: string; content: string } | { kind: "steer"; ts: string; content: string } @@ -102,6 +118,7 @@ type TurnType = inputTokens: number; outputTokens: number; toolCallCount: number | null; + reasoning: TurnReasoning | null; } | { kind: "tool"; @@ -277,6 +294,17 @@ interface PendingCommand { script: string; } +function readTurnReasoning(props: UnknownRecord): TurnReasoning | null { + const reasoning = getObject(props, "reasoning"); + if (!reasoning) return null; + // getString treats "" as absent, so a provider that sends an empty field + // reads the same as one that sends nothing. + const summary = getString(reasoning, "summary") ?? null; + const trace = getString(reasoning, "trace") ?? null; + if (!summary && !trace) return null; + return { summary, trace }; +} + export function buildStageActivity( events: EventEnvelope[], stageId: string, @@ -320,6 +348,7 @@ export function buildStageActivity( inputTokens: getNumber(billing, "input_tokens") ?? 0, outputTokens: getNumber(billing, "output_tokens") ?? 0, toolCallCount: getNumber(props, "tool_call_count") ?? null, + reasoning: readTurnReasoning(props), }); break; } @@ -333,6 +362,8 @@ export function buildStageActivity( inputTokens: getNumber(billing, "input_tokens") ?? 0, outputTokens: getNumber(billing, "output_tokens") ?? 0, toolCallCount: null, + // Only agent.message carries reasoning; prompt stages have none. + reasoning: null, }); } break; @@ -1146,7 +1177,44 @@ function ToolGroupRow({ ); } -function EventDetails({ +const REASONING_PREVIEW_CHARS = 280; + +/** + * Reasoning is raw model output, not authored Markdown, so it renders as + * preformatted text: a trace's own line breaks are part of what it says, and + * parsing it as Markdown would eat them along with any leading `#` or `-`. + */ +function CollapsibleText({ text }: { text: string }) { + const [expanded, setExpanded] = useState(false); + const contentId = useId(); + + if (text.length <= REASONING_PREVIEW_CHARS) { + return

{text}

; + } + + return ( +
+

+ {expanded + ? text + : `${text.slice(0, REASONING_PREVIEW_CHARS).trimEnd()}…`} +

+ +
+ ); +} + +export function EventDetails({ turn, runStart, hideMeta = false, @@ -1191,6 +1259,23 @@ function EventDetails({ {turnSummary(turn)} )} + {/* + Reasoning follows the message rather than preceding it, the way it + ran: a trace can be thousands of characters, and leading with one + would push the answer the user clicked on below the fold. + */} + {turn.reasoning?.summary && ( + + + + )} + {turn.reasoning?.trace && ( + + + + )} {turn.toolCallCount != null && turn.toolCallCount > 0 && ( {turn.toolCallCount} From 3f2bb4f4b8a051a1ce784d8d82786208bf142368 Mon Sep 17 00:00:00 2001 From: Release Repro Date: Sat, 25 Jul 2026 11:10:06 -0400 Subject: [PATCH 6/6] Simplify stage detail reasoning UI --- .../app/routes/run-stages-chat.test.tsx | 4 + .../app/routes/run-stages-details.test.tsx | 12 +- apps/fabro-web/app/routes/run-stages.test.ts | 17 +- apps/fabro-web/app/routes/run-stages.tsx | 170 +++++++++--------- 4 files changed, 107 insertions(+), 96 deletions(-) diff --git a/apps/fabro-web/app/routes/run-stages-chat.test.tsx b/apps/fabro-web/app/routes/run-stages-chat.test.tsx index be86b89fd..c4307ebbc 100644 --- a/apps/fabro-web/app/routes/run-stages-chat.test.tsx +++ b/apps/fabro-web/app/routes/run-stages-chat.test.tsx @@ -33,6 +33,8 @@ describe("StageChatView", () => { content: "Finished", inputTokens: 0, outputTokens: 0, + toolCallCount: null, + reasoning: null, }, { kind: "tool", @@ -72,6 +74,7 @@ describe("StageChatView", () => { inputTokens: 0, outputTokens: 0, toolCallCount: 1, + reasoning: null, }, { kind: "tool", @@ -89,6 +92,7 @@ describe("StageChatView", () => { inputTokens: 10, outputTokens: 20, toolCallCount: null, + reasoning: null, }, ]} pendingTools={[]} diff --git a/apps/fabro-web/app/routes/run-stages-details.test.tsx b/apps/fabro-web/app/routes/run-stages-details.test.tsx index d2cd6ad84..da18f7e67 100644 --- a/apps/fabro-web/app/routes/run-stages-details.test.tsx +++ b/apps/fabro-web/app/routes/run-stages-details.test.tsx @@ -1,11 +1,13 @@ import { describe, expect, test } from "bun:test"; import { renderToStaticMarkup } from "react-dom/server"; -import { EventDetails, type TurnReasoning } from "./run-stages"; +import type { ReasoningOutput } from "@qltysh/fabro-api-client"; + +import { EventDetails } from "./run-stages"; const RUN_START = "2026-04-09T12:00:00Z"; -function assistantMarkup(reasoning: TurnReasoning | null): string { +function assistantMarkup(reasoning: ReasoningOutput | null): string { return renderToStaticMarkup( { }); test("labels a trace-only response Reasoning, not Reasoning trace", () => { - const html = assistantMarkup({ summary: null, trace: "Considered A." }); + const html = assistantMarkup({ trace: "Considered A." }); expect(html).toContain("Reasoning"); expect(html).not.toContain("Reasoning trace"); @@ -50,7 +52,7 @@ describe("EventDetails reasoning", () => { }); test("renders short reasoning in full, with no disclosure control", () => { - const html = assistantMarkup({ summary: null, trace: "Considered A." }); + const html = assistantMarkup({ trace: "Considered A." }); expect(html).not.toContain("Show all"); expect(html).not.toContain("aria-expanded"); @@ -58,7 +60,7 @@ describe("EventDetails reasoning", () => { test("connects a long trace's expand button to its controlled content", () => { const trace = "x".repeat(281); - const html = assistantMarkup({ summary: null, trace }); + const html = assistantMarkup({ trace }); const controls = html.match(/aria-controls="([^"]+)"/)?.[1]; expect(controls).toBeDefined(); diff --git a/apps/fabro-web/app/routes/run-stages.test.ts b/apps/fabro-web/app/routes/run-stages.test.ts index c77bf73fb..31a46eaef 100644 --- a/apps/fabro-web/app/routes/run-stages.test.ts +++ b/apps/fabro-web/app/routes/run-stages.test.ts @@ -499,9 +499,15 @@ describe("eventsToActivity", () => { ).toEqual({ summary: "Checked the config", trace: "step one…" }); // Anthropic thinking arrives as a trace with no summary. - expect(reasoningOf({ text: "Done.", reasoning: { trace: "step one…" } })).toEqual( - { summary: null, trace: "step one…" }, - ); + expect( + reasoningOf({ text: "Done.", reasoning: { trace: "step one…" } }), + ).toEqual({ trace: "step one…" }); + expect( + reasoningOf({ + text: "Done.", + reasoning: { summary: "Checked the config" }, + }), + ).toEqual({ summary: "Checked the config" }); expect(reasoningOf({ text: "Done." })).toBe(null); // A provider that sends the key but nothing usable reads as "none". @@ -780,6 +786,7 @@ describe("groupConsecutiveTools", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }; const c = toolTurn({ ts: "2026-04-09T12:00:03Z", toolName: "shell" }); const result = groupConsecutiveTools([ @@ -857,6 +864,8 @@ describe("buildChatItems", () => { content, inputTokens: 0, outputTokens: 0, + toolCallCount: null, + reasoning: null, }; } @@ -1074,6 +1083,7 @@ describe("buildThreadDnaItems", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, selection: { kind: "single" as const, turnIndex }, }; @@ -1295,6 +1305,7 @@ describe("tool-call-only agent responses", () => { inputTokens: 0, outputTokens: 0, toolCallCount: 3, + reasoning: null, }; const withOneTool = { ...withTools, toolCallCount: 1 }; const withoutCount = { ...withTools, toolCallCount: null }; diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index 9357c17c7..43f4410f1 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -1,4 +1,10 @@ -import { useId, useMemo, useReducer, useState } from "react"; +import { + useId, + useMemo, + useReducer, + useState, + type ReactNode, +} from "react"; import { Link, useParams } from "react-router"; import { ArrowDownTrayIcon, @@ -88,23 +94,13 @@ import { } from "../lib/unknown"; import type { EventEnvelope, + ReasoningOutput, StageHandler, StageModelUsage, } from "@qltysh/fabro-api-client"; export const handle = { wide: true, fullHeight: true }; -/** - * Readable reasoning a provider disclosed for one model response: `summary` is - * the model's own recap, `trace` its verbatim reasoning. Providers send one, - * the other, or both, and opaque material (signatures, redacted blocks) never - * reaches the wire, so an absent field means nothing was disclosed. - */ -export interface TurnReasoning { - summary: string | null; - trace: string | null; -} - type TurnType = | { kind: "system"; ts: string; content: string } | { kind: "steer"; ts: string; content: string } @@ -118,7 +114,7 @@ type TurnType = inputTokens: number; outputTokens: number; toolCallCount: number | null; - reasoning: TurnReasoning | null; + reasoning: ReasoningOutput | null; } | { kind: "tool"; @@ -294,15 +290,15 @@ interface PendingCommand { script: string; } -function readTurnReasoning(props: UnknownRecord): TurnReasoning | null { +function readTurnReasoning(props: UnknownRecord): ReasoningOutput | null { const reasoning = getObject(props, "reasoning"); if (!reasoning) return null; // getString treats "" as absent, so a provider that sends an empty field // reads the same as one that sends nothing. const summary = getString(reasoning, "summary") ?? null; const trace = getString(reasoning, "trace") ?? null; - if (!summary && !trace) return null; - return { summary, trace }; + if (summary) return trace ? { summary, trace } : { summary }; + return trace ? { trace } : null; } export function buildStageActivity( @@ -1177,39 +1173,60 @@ function ToolGroupRow({ ); } -const REASONING_PREVIEW_CHARS = 280; +const COLLAPSIBLE_PREVIEW_CHARS = 280; /** - * Reasoning is raw model output, not authored Markdown, so it renders as - * preformatted text: a trace's own line breaks are part of what it says, and - * parsing it as Markdown would eat them along with any leading `#` or `-`. + * Shared disclosure for long stage text. By default it preserves raw text; + * callers may supply a full-content renderer for authored formats such as + * Markdown while retaining the same plain-text preview and accessible toggle. */ -function CollapsibleText({ text }: { text: string }) { +function CollapsibleContent({ + text, + className = "", + textClassName = "", + renderFull, +}: { + text: string; + className?: string; + textClassName?: string; + renderFull?: (text: string) => ReactNode; +}) { const [expanded, setExpanded] = useState(false); const contentId = useId(); - - if (text.length <= REASONING_PREVIEW_CHARS) { - return

{text}

; - } + const isLong = text.length > COLLAPSIBLE_PREVIEW_CHARS; + const preview = isLong + ? `${text.slice(0, COLLAPSIBLE_PREVIEW_CHARS).trimEnd()}…` + : text; return ( -
-

- {expanded - ? text - : `${text.slice(0, REASONING_PREVIEW_CHARS).trimEnd()}…`} -

- +
+ {/* + `w-full` is load-bearing for ChatUserCard. Its `w-fit max-w-[85%]` + bubble is measured intrinsically before being clamped, so this wrapper + must fill the resolved width to keep prompt text inside the bubble. + */} +
+ {renderFull && (!isLong || expanded) ? ( + renderFull(text) + ) : ( +

{expanded ? text : preview}

+ )} +
+ {isLong && ( + + )}
); } @@ -1231,6 +1248,9 @@ export function EventDetails({ })(); const assistantContent = turn.kind === "assistant" ? nonBlankAssistantContent(turn) : null; + const reasoning = turn.kind === "assistant" ? turn.reasoning : null; + const reasoningSummary = + reasoning && "summary" in reasoning ? reasoning.summary : null; return (
@@ -1264,16 +1284,22 @@ export function EventDetails({ ran: a trace can be thousands of characters, and leading with one would push the answer the user clicked on below the fold. */} - {turn.reasoning?.summary && ( + {reasoningSummary && ( - + )} - {turn.reasoning?.trace && ( + {reasoning?.trace && ( - + )} {turn.toolCallCount != null && turn.toolCallCount > 0 && ( @@ -1555,49 +1581,17 @@ function ToolGroupChildRow({ ); } -const CHAT_PROMPT_PREVIEW_CHARS = 280; - // User-side bubble. The stage prompt (and any steer / pair-user message over // the preview limit) collapses to a preview with an expand toggle; expanded // content renders as markdown. function ChatUserCard({ content }: { content: string }) { - const [expanded, setExpanded] = useState(false); - const contentId = useId(); - const isLong = content.length > CHAT_PROMPT_PREVIEW_CHARS; return ( -
- {/* - `w-full` is load-bearing. `items-start` keeps the expand button hugging - its label, but it also leaves this wrapper intrinsically sized, and the - bubble's own `w-fit` width is only clamped to `max-w-[85%]` after that - intrinsic pass. The text then keeps the wider pre-clamp measurement and - spills past the bubble. Filling the resolved width sidesteps it. - */} -
- {expanded ? ( - - ) : ( -

- {isLong - ? `${content.slice(0, CHAT_PROMPT_PREVIEW_CHARS).trimEnd()}…` - : content} -

- )} -
- {isLong && ( - - )} -
+ } + /> ); } @@ -1678,7 +1672,7 @@ export function StageChatView({

); case "assistant": { - const hasText = turn.content.trim().length > 0; + const assistantContent = nonBlankAssistantContent(turn); const metric = turnMetric(turn); const isFinal = turnIndex === lastAssistantTurnIndex && !stageActive; @@ -1688,10 +1682,10 @@ export function StageChatView({ // separate chips. It has nothing to show, and an empty node would // still take a slot in this gap-4 column, doubling the space // between the chips on either side of it. - if (!hasText && !showFooter) return null; + if (!assistantContent && !showFooter) return null; return (
- {hasText && } + {assistantContent && } {showFooter && (
{metric && {metric}}