diff --git a/apps/fabro-web/app/components/stage-inference-indicator.test.tsx b/apps/fabro-web/app/components/stage-inference-indicator.test.tsx deleted file mode 100644 index 45da56ae4..000000000 --- a/apps/fabro-web/app/components/stage-inference-indicator.test.tsx +++ /dev/null @@ -1,107 +0,0 @@ -import { afterEach, beforeEach, describe, expect, test } from "bun:test"; -import TestRenderer, { act } from "react-test-renderer"; - -import { LlmOutputKind } from "@qltysh/fabro-api-client"; -import type { StageInferenceProjection } from "@qltysh/fabro-api-client"; - -import { StageInferenceIndicator } from "./stage-inference-indicator"; - -const OPENED_AT = new Date(Date.now() - 12_000).toISOString(); - -function makeInference( - overrides: Partial = {}, -): StageInferenceProjection { - return { - session_id: "ses_root", - started_at: OPENED_AT, - requested_model: { - provider: "anthropic", - model_id: "claude-fable-5", - }, - retries: 0, - ...overrides, - }; -} - -function render( - inference: StageInferenceProjection | null | undefined, - settled = false, -): string { - let renderer!: TestRenderer.ReactTestRenderer; - act(() => { - renderer = TestRenderer.create( - , - ); - }); - const output = JSON.stringify(renderer.toJSON()); - act(() => renderer.unmount()); - return output; -} - -describe("StageInferenceIndicator", () => { - const actGlobal = globalThis as { - IS_REACT_ACT_ENVIRONMENT?: boolean; - }; - const previousActEnvironment = actGlobal.IS_REACT_ACT_ENVIRONMENT; - - beforeEach(() => { - actGlobal.IS_REACT_ACT_ENVIRONMENT = true; - }); - - afterEach(() => { - if (previousActEnvironment === undefined) { - delete actGlobal.IS_REACT_ACT_ENVIRONMENT; - } else { - actGlobal.IS_REACT_ACT_ENVIRONMENT = previousActEnvironment; - } - }); - - test("renders nothing without an open bracket", () => { - expect(render(undefined)).toBe("null"); - expect(render(null)).toBe("null"); - }); - - test("names the requested model while nothing has come back", () => { - const output = render(makeInference()); - expect(output).toContain("Model request"); - expect(output).toContain("waiting on claude-fable-5"); - expect(output).toContain('"aria-live":"polite"'); - expect(output).toContain('"aria-hidden":"true"'); - // No completion estimate exists, so none may be shown. - expect(output).not.toContain("%"); - }); - - test("reports the observed first-output kind", () => { - expect( - render(makeInference({ first_output_kind: LlmOutputKind.REASONING })), - ).toContain("reasoning"); - expect( - render(makeInference({ first_output_kind: LlmOutputKind.TEXT })), - ).toContain("writing"); - expect( - render(makeInference({ first_output_kind: LlmOutputKind.TOOL_CALL })), - ).toContain("calling tools"); - }); - - test("never says thinking for non-reasoning output", () => { - for (const kind of [LlmOutputKind.TEXT, LlmOutputKind.TOOL_CALL]) { - expect(render(makeInference({ first_output_kind: kind }))).not.toContain( - "thinking", - ); - } - }); - - test("counts retries without presenting them as failure", () => { - const output = render(makeInference({ retries: 2 })); - expect(output).toContain("retry 2"); - expect(output).not.toContain("failed"); - }); - - test("goes static once the run can no longer advance the bracket", () => { - const output = render(makeInference(), true); - // An open bracket on a settled run means we never learned how the request - // ended — animating it would claim work that may not be happening. - expect(output).toContain("never completed"); - expect(output).not.toContain("animate-pulse"); - }); -}); diff --git a/apps/fabro-web/app/components/stage-inference-indicator.tsx b/apps/fabro-web/app/components/stage-inference-indicator.tsx deleted file mode 100644 index a329f3fa0..000000000 --- a/apps/fabro-web/app/components/stage-inference-indicator.tsx +++ /dev/null @@ -1,87 +0,0 @@ -import { LlmOutputKind } from "@qltysh/fabro-api-client"; -import type { StageInferenceProjection } from "@qltysh/fabro-api-client"; - -import { Tooltip } from "./ui"; -import { formatAbsoluteTs, formatDurationSecs } from "../lib/format"; -import { elapsedSecsSince, useTickingNow } from "../lib/time"; - -export interface StageInferenceIndicatorProps { - /** Open inference bracket from the stage projection, if there is one. */ - inference: StageInferenceProjection | null | undefined; - /** - * The run can no longer make progress on this bracket: it reached a terminal - * status, or the stall watchdog fired. An open bracket then means *we never - * learned how the request ended*, not *it is still working*, so the readout - * goes static. - */ - settled: boolean; -} - -const ACTIVITY_LABEL: Record = { - [LlmOutputKind.REASONING]: "reasoning", - [LlmOutputKind.TEXT]: "writing", - [LlmOutputKind.TOOL_CALL]: "calling tools", -}; - -/** - * Live readout for an open model request. - * - * Says only what the event log proves. There is no progress bar, percentage, - * or ETA, because no completion estimate exists; the elapsed clock counts - * since the request opened rather than claiming the model is still working; - * and "reasoning" appears only when the provider actually sent reasoning - * output, never as a guess filling a gap in the log. - */ -export function StageInferenceIndicator({ - inference, - settled, -}: StageInferenceIndicatorProps) { - // Ticking is what distinguishes "we are still hearing from this request" - // from "this is a record of one that never closed", so it stops the moment - // the bracket can no longer advance. - const now = useTickingNow(Boolean(inference) && !settled); - - if (!inference) return null; - - if (settled) { - return ( -

- Model request opened {formatAbsoluteTs(inference.started_at)}, never - completed -

- ); - } - - const elapsedSecs = elapsedSecsSince(inference.started_at, now); - const activity = inference.first_output_kind - ? ACTIVITY_LABEL[inference.first_output_kind] - : `waiting on ${inference.requested_model.model_id}`; - - const statusParts = ["Model request", activity]; - // A retry that later succeeds is normal, so this is a count, not a failure. - if (inference.retries > 0) { - statusParts.push(`retry ${inference.retries}`); - } - - return ( -

- - - - -

- ); -} diff --git a/apps/fabro-web/app/lib/run-events.test.tsx b/apps/fabro-web/app/lib/run-events.test.tsx index 11544b14d..e984be74d 100644 --- a/apps/fabro-web/app/lib/run-events.test.tsx +++ b/apps/fabro-web/app/lib/run-events.test.tsx @@ -150,7 +150,7 @@ describe("queryKeysForRunEvent", () => { ]); }); - test("watchdog timeout refreshes the stage events that settle inference", () => { + test("watchdog timeout refreshes the stage events for that stage", () => { expect( queryKeysForRunEvent("run-1", "watchdog.timeout", "code@1"), ).toEqual([queryKeys.runs.stageEvents("run-1", "code@1")]); diff --git a/apps/fabro-web/app/routes/run-stages-chat.test.tsx b/apps/fabro-web/app/routes/run-stages-chat.test.tsx index 9ca84db99..c4307ebbc 100644 --- a/apps/fabro-web/app/routes/run-stages-chat.test.tsx +++ b/apps/fabro-web/app/routes/run-stages-chat.test.tsx @@ -33,6 +33,8 @@ describe("StageChatView", () => { content: "Finished", inputTokens: 0, outputTokens: 0, + toolCallCount: null, + reasoning: null, }, { kind: "tool", @@ -52,6 +54,60 @@ describe("StageChatView", () => { expect(html).toContain("1m 12s"); }); + test("renders no node for a text-free assistant turn between tool batches", () => { + const html = renderToStaticMarkup( + , + ); + + // The boundary keeps the two batches as separate chips, but contributes + // no element of its own between them. + expect(html.match(/1 tool call/g)).toHaveLength(2); + expect(html).not.toContain('class="prose prose-sm max-w-none"'); + expect(html.match(/class="prose /g)).toHaveLength(1); + expect(html).toContain("Done"); + }); + test("connects a long prompt's expand button to its controlled content", () => { const html = renderToStaticMarkup( , + ); +} + +describe("EventDetails reasoning", () => { + test("shows nothing when the response disclosed no reasoning", () => { + const html = assistantMarkup(null); + + expect(html).toContain("Refactored the auth module."); + expect(html).not.toContain("Reasoning"); + }); + + test("labels a trace-only response Reasoning, not Reasoning trace", () => { + const html = assistantMarkup({ trace: "Considered A." }); + + expect(html).toContain("Reasoning"); + expect(html).not.toContain("Reasoning trace"); + expect(html).toContain("Considered A."); + }); + + test("distinguishes the summary from the verbatim trace when both arrive", () => { + const html = assistantMarkup({ + summary: "Checked the config.", + trace: "Considered A.", + }); + + expect(html).toContain("Reasoning trace"); + expect(html).toContain("Checked the config."); + expect(html).toContain("Considered A."); + }); + + test("renders short reasoning in full, with no disclosure control", () => { + const html = assistantMarkup({ trace: "Considered A." }); + + expect(html).not.toContain("Show all"); + expect(html).not.toContain("aria-expanded"); + }); + + test("connects a long trace's expand button to its controlled content", () => { + const trace = "x".repeat(281); + const html = assistantMarkup({ trace }); + + const controls = html.match(/aria-controls="([^"]+)"/)?.[1]; + expect(controls).toBeDefined(); + expect(html).toContain(`id="${controls}"`); + expect(html).toContain('aria-expanded="false"'); + expect(html).toContain("Show all (281 characters)"); + // Collapsed, so the preview is truncated rather than the whole trace. + expect(html).not.toContain(trace); + expect(html).toContain(`${"x".repeat(280)}…`); + }); +}); diff --git a/apps/fabro-web/app/routes/run-stages.test.ts b/apps/fabro-web/app/routes/run-stages.test.ts index 2238aaf74..31a46eaef 100644 --- a/apps/fabro-web/app/routes/run-stages.test.ts +++ b/apps/fabro-web/app/routes/run-stages.test.ts @@ -114,6 +114,7 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, ]); @@ -131,6 +132,7 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, ]); }); @@ -390,6 +392,7 @@ describe("eventsToActivity", () => { inputTokens: 120, outputTokens: 30, toolCallCount: null, + reasoning: null, }, ]); }); @@ -438,6 +441,7 @@ describe("eventsToActivity", () => { inputTokens: 10, outputTokens: 5, toolCallCount: null, + reasoning: null, }, ]); }); @@ -465,10 +469,54 @@ describe("eventsToActivity", () => { inputTokens: 0, outputTokens: 4, toolCallCount: null, + reasoning: null, }, ]); }); + test("reads disclosed reasoning off agent.message", () => { + function reasoningOf(properties: Record) { + const turns = eventsToActivity( + [ + envelope(1, { + event: "agent.message", + stage_id: "plan@1", + node_id: "plan", + properties, + }), + ], + "plan@1", + ); + expect(turns[0].kind).toBe("assistant"); + return turns[0].kind === "assistant" ? turns[0].reasoning : undefined; + } + + expect( + reasoningOf({ + text: "Done.", + reasoning: { summary: "Checked the config", trace: "step one…" }, + }), + ).toEqual({ summary: "Checked the config", trace: "step one…" }); + + // Anthropic thinking arrives as a trace with no summary. + expect( + reasoningOf({ text: "Done.", reasoning: { trace: "step one…" } }), + ).toEqual({ trace: "step one…" }); + expect( + reasoningOf({ + text: "Done.", + reasoning: { summary: "Checked the config" }, + }), + ).toEqual({ summary: "Checked the config" }); + + expect(reasoningOf({ text: "Done." })).toBe(null); + // A provider that sends the key but nothing usable reads as "none". + expect(reasoningOf({ text: "Done.", reasoning: {} })).toBe(null); + expect( + reasoningOf({ text: "Done.", reasoning: { summary: "", trace: "" } }), + ).toBe(null); + }); + test("formatStageModelUsageLabel includes reasoning effort when present", () => { expect( formatStageModelUsageLabel({ @@ -738,6 +786,7 @@ describe("groupConsecutiveTools", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }; const c = toolTurn({ ts: "2026-04-09T12:00:03Z", toolName: "shell" }); const result = groupConsecutiveTools([ @@ -815,6 +864,8 @@ describe("buildChatItems", () => { content, inputTokens: 0, outputTokens: 0, + toolCallCount: null, + reasoning: null, }; } @@ -884,19 +935,6 @@ describe("buildChatItems", () => { }); describe("buildStageActivity pending tools", () => { - test("records watchdog settlement only for the selected stage", () => { - const events: EventEnvelope[] = [ - envelope(1, { - event: "watchdog.timeout", - stage_id: "plan@1", - node_id: "plan", - }), - ]; - - expect(buildStageActivity(events, "plan@1").watchdogTimedOut).toBe(true); - expect(buildStageActivity(events, "code@1").watchdogTimedOut).toBe(false); - }); - test("returns started-but-not-completed calls for the stage", () => { const events: EventEnvelope[] = [ envelope(1, { @@ -1045,6 +1083,7 @@ describe("buildThreadDnaItems", () => { inputTokens: 0, outputTokens: 0, toolCallCount: null, + reasoning: null, }, selection: { kind: "single" as const, turnIndex }, }; @@ -1229,6 +1268,7 @@ describe("tool-call-only agent responses", () => { inputTokens: 4200, outputTokens: 96, toolCallCount: 2, + reasoning: null, }, ]); }); @@ -1265,6 +1305,7 @@ describe("tool-call-only agent responses", () => { inputTokens: 0, outputTokens: 0, toolCallCount: 3, + reasoning: null, }; const withOneTool = { ...withTools, toolCallCount: 1 }; const withoutCount = { ...withTools, toolCallCount: null }; diff --git a/apps/fabro-web/app/routes/run-stages.tsx b/apps/fabro-web/app/routes/run-stages.tsx index 48111ce23..43f4410f1 100644 --- a/apps/fabro-web/app/routes/run-stages.tsx +++ b/apps/fabro-web/app/routes/run-stages.tsx @@ -1,4 +1,10 @@ -import { useId, useMemo, useReducer, useState } from "react"; +import { + useId, + useMemo, + useReducer, + useState, + type ReactNode, +} from "react"; import { Link, useParams } from "react-router"; import { ArrowDownTrayIcon, @@ -31,7 +37,6 @@ import type { ThreadDnaSelection, } from "../components/event-debug"; import { StageContext } from "../components/stage-context"; -import { StageInferenceIndicator } from "../components/stage-inference-indicator"; import { StageInsightsSidebar } from "../components/stage-insights-sidebar"; import { StageSidebar } from "../components/stage-sidebar"; import type { Stage } from "../components/stage-sidebar"; @@ -73,7 +78,6 @@ import { useRunStages, useRunState, } from "../lib/queries"; -import { isTerminalRunStatus } from "../lib/run-actions"; import { STAGE_ACTIVITY_EVENT_TYPES, type StageActivityEventType, @@ -82,11 +86,16 @@ import { ACTIVE_STAGE_STATES, mapRunStagesToSidebarStages, } from "../lib/stage-sidebar"; -import { getNumber, getString, type UnknownRecord } from "../lib/unknown"; +import { + getNumber, + getObject, + getString, + type UnknownRecord, +} from "../lib/unknown"; import type { EventEnvelope, + ReasoningOutput, StageHandler, - StageInferenceProjection, StageModelUsage, } from "@qltysh/fabro-api-client"; @@ -105,6 +114,7 @@ type TurnType = inputTokens: number; outputTokens: number; toolCallCount: number | null; + reasoning: ReasoningOutput | null; } | { kind: "tool"; @@ -181,7 +191,9 @@ type StageActivityAction = | { type: "searchChanged"; search: string }; const initialStageActivityState = (): StageActivityState => ({ - tab: "primary", + // Only agent stages offer "chat"; every other renderer resolves this to + // "primary" through `availableTabs`, so this is the default for both. + tab: "chat", selectedKinds: [...EVENT_KINDS], selectedDebugCategories: [], search: "", @@ -271,7 +283,6 @@ export interface PendingToolCall { interface StageActivity { turns: TurnType[]; pendingTools: PendingToolCall[]; - watchdogTimedOut: boolean; } interface PendingCommand { @@ -279,6 +290,17 @@ interface PendingCommand { script: string; } +function readTurnReasoning(props: UnknownRecord): ReasoningOutput | null { + const reasoning = getObject(props, "reasoning"); + if (!reasoning) return null; + // getString treats "" as absent, so a provider that sends an empty field + // reads the same as one that sends nothing. + const summary = getString(reasoning, "summary") ?? null; + const trace = getString(reasoning, "trace") ?? null; + if (summary) return trace ? { summary, trace } : { summary }; + return trace ? { trace } : null; +} + export function buildStageActivity( events: EventEnvelope[], stageId: string, @@ -287,17 +309,12 @@ export function buildStageActivity( const pendingTools = new Map(); let pendingCommand: PendingCommand | undefined; let sawAssistantMessage = false; - let watchdogTimedOut = false; for (const e of events) { const eventName = e.event; if (activityEventStageId(e) !== stageId) { continue; } - if (eventName === "watchdog.timeout") { - watchdogTimedOut = true; - continue; - } if ( !eventName || !STAGE_ACTIVITY_EVENT_SET.has(eventName) @@ -327,6 +344,7 @@ export function buildStageActivity( inputTokens: getNumber(billing, "input_tokens") ?? 0, outputTokens: getNumber(billing, "output_tokens") ?? 0, toolCallCount: getNumber(props, "tool_call_count") ?? null, + reasoning: readTurnReasoning(props), }); break; } @@ -340,6 +358,8 @@ export function buildStageActivity( inputTokens: getNumber(billing, "input_tokens") ?? 0, outputTokens: getNumber(billing, "output_tokens") ?? 0, toolCallCount: null, + // Only agent.message carries reasoning; prompt stages have none. + reasoning: null, }); } break; @@ -452,7 +472,6 @@ export function buildStageActivity( return { turns, - watchdogTimedOut, pendingTools: Array.from(pendingTools, ([toolCallId, tool]) => ({ toolCallId, toolName: tool.toolName, @@ -1154,7 +1173,65 @@ function ToolGroupRow({ ); } -function EventDetails({ +const COLLAPSIBLE_PREVIEW_CHARS = 280; + +/** + * Shared disclosure for long stage text. By default it preserves raw text; + * callers may supply a full-content renderer for authored formats such as + * Markdown while retaining the same plain-text preview and accessible toggle. + */ +function CollapsibleContent({ + text, + className = "", + textClassName = "", + renderFull, +}: { + text: string; + className?: string; + textClassName?: string; + renderFull?: (text: string) => ReactNode; +}) { + const [expanded, setExpanded] = useState(false); + const contentId = useId(); + const isLong = text.length > COLLAPSIBLE_PREVIEW_CHARS; + const preview = isLong + ? `${text.slice(0, COLLAPSIBLE_PREVIEW_CHARS).trimEnd()}…` + : text; + + return ( +
+ {/* + `w-full` is load-bearing for ChatUserCard. Its `w-fit max-w-[85%]` + bubble is measured intrinsically before being clamped, so this wrapper + must fill the resolved width to keep prompt text inside the bubble. + */} +
+ {renderFull && (!isLong || expanded) ? ( + renderFull(text) + ) : ( +

{expanded ? text : preview}

+ )} +
+ {isLong && ( + + )} +
+ ); +} + +export function EventDetails({ turn, runStart, hideMeta = false, @@ -1171,6 +1248,9 @@ function EventDetails({ })(); const assistantContent = turn.kind === "assistant" ? nonBlankAssistantContent(turn) : null; + const reasoning = turn.kind === "assistant" ? turn.reasoning : null; + const reasoningSummary = + reasoning && "summary" in reasoning ? reasoning.summary : null; return (
@@ -1199,6 +1279,29 @@ function EventDetails({ {turnSummary(turn)} )} + {/* + Reasoning follows the message rather than preceding it, the way it + ran: a trace can be thousands of characters, and leading with one + would push the answer the user clicked on below the fold. + */} + {reasoningSummary && ( + + + + )} + {reasoning?.trace && ( + + + + )} {turn.toolCallCount != null && turn.toolCallCount > 0 && ( {turn.toolCallCount} @@ -1478,42 +1581,17 @@ function ToolGroupChildRow({ ); } -const CHAT_PROMPT_PREVIEW_CHARS = 280; - // User-side bubble. The stage prompt (and any steer / pair-user message over // the preview limit) collapses to a preview with an expand toggle; expanded // content renders as markdown. function ChatUserCard({ content }: { content: string }) { - const [expanded, setExpanded] = useState(false); - const contentId = useId(); - const isLong = content.length > CHAT_PROMPT_PREVIEW_CHARS; return ( -
-
- {expanded ? ( - - ) : ( -

- {isLong - ? `${content.slice(0, CHAT_PROMPT_PREVIEW_CHARS).trimEnd()}…` - : content} -

- )} -
- {isLong && ( - - )} -
+ } + /> ); } @@ -1594,13 +1672,21 @@ export function StageChatView({

); case "assistant": { + const assistantContent = nonBlankAssistantContent(turn); const metric = turnMetric(turn); const isFinal = turnIndex === lastAssistantTurnIndex && !stageActive; + const showFooter = isFinal && Boolean(metric || duration); + // A text-free assistant turn is the boundary between two batches + // of tool calls, kept in the turn stream so those batches stay + // separate chips. It has nothing to show, and an empty node would + // still take a slot in this gap-4 column, doubling the space + // between the chips on either side of it. + if (!assistantContent && !showFooter) return null; return (
- - {isFinal && (metric || duration) && ( + {assistantContent && } + {showFooter && (
{metric && {metric}} {duration && {duration}} @@ -2075,8 +2161,6 @@ function RunStageActivityStage({ selectedStage, stages, runStart, - inference, - runSettled, tab, selectedKinds, selectedDebugCategories, @@ -2090,8 +2174,6 @@ function RunStageActivityStage({ selectedStage: Stage; stages: Stage[]; runStart: string | undefined; - inference: StageInferenceProjection | null | undefined; - runSettled: boolean; tab: EventsTab; selectedKinds: EventKind[]; selectedDebugCategories: DebugCategory[]; @@ -2103,15 +2185,11 @@ function RunStageActivityStage({ }) { const selectedStageId = selectedStage.id; const stageEventsQuery = useRunStageEvents(runId, selectedStageId); - // An open bracket on a run that can no longer advance means we never learned - // how the request ended, not that it is still working. The watchdog stays - // the authority on "actually stuck", so its timeout settles the readout too. const activity = useMemo( () => buildStageActivity(stageEventsQuery.data ?? [], selectedStageId), [stageEventsQuery.data, selectedStageId], ); const { turns } = activity; - const inferenceSettled = runSettled || activity.watchdogTimedOut; const renderer: StageRenderer = selectStageRenderer(selectedStage.handler); const debugEvents = useMemo(() => { return (stageEventsQuery.data ?? []).filter( @@ -2259,11 +2337,6 @@ function RunStageActivityStage({

)} - -
);