fix(lens): make tool steps and conversations readable (#44645)

* feat(lens): add per-trace review models to jobs and progress

* feat(lens): append worker reviews to the job, capped, and count every review

* feat(lens): report a review with reasoning for each screened trace

* chore(ui): regenerate api types for lens job reviews

* feat(lens): type job reviews and fill them in lens fixtures

* feat(lens): add live review playback model

* feat(lens): pick the analysis model and slow single-review pacing

* feat(lens): add sample reviews for previewing the live run

* feat(lens): add live run layout with queue, reading trace and conclusions

* feat(lens): show the live run on investigations and open it from run now

* feat(lens): stream large review backlogs at 150ms or less and list newest first

* fix(lens): show the live run only for real reviews and keep fixtures test-only

* refactor(lens): restyle the live run as the native progress panel

* fix(lens): retry contended investigation updates with jittered backoff

* feat(lens): add a reading ticker line and replay for finished runs

* feat(lens): collapse the live run to an ambient line with show work

* fix(ui): crop the cerebras logo viewBox to its mark so it reads at icon size

* feat(lens): format review span previews as readable messages

* feat(lens): derive strip status, honest issue counts and drawer focus from a job

* feat(lens): track active jobs before their first review

* feat(lens): add a live trace results drawer with readable spans

* feat(lens): put the live strip under the progress bar and drop the inline panel

* feat(lens): add an ambient live strip that opens the drawer

* fix(lens): wait out provider rate limits and retry model calls four times

* style(lens): format repository contention tests

* feat(lens): read review spans as a conversation timeline

Turns spans into the user's ask, tool calls with args and results, and the agent's reply, dropping system prompts. Also handles a preview cut that lands inside the Output header.

* fix(lens): list recorded agents in the run now dialog

The run now agent field used a native datalist, whose suggestions do not show inside the modal dialog, so the agent list looked empty even though /lens/agents returned names. Use the same Combobox as investigation setup.

* feat(lens): pace live playback so each trace stays readable

Every trace now stays up for at least 1.5s. A backlog is cleared by skipping to the newest few instead of flickering through them. Conclusions count traces per check and kind, and new helpers cover share bars, group filters and flashes.

* feat(lens): keep the live run ambient until View run is clicked

The drawer no longer opens on Run or when entering a running investigation. LiveRun takes reviews as a prop so it can move to a dedicated reviews endpoint.

* feat(lens): show the live run as a two-pane trace and conclusions view

Left pane: the trace being reviewed as a readable timeline, followed by Lens's reasoning and the verdict. Right pane: ranked conclusion groups with share bars, plus a trace list you can filter.

* fix(lens): run several investigations per worker and poll every two seconds

* feat(lens): add worker slot and poll interval settings

* feat(lens): add list summaries and an incremental review filter

* perf(lens): strip reviews and run attributes from the lens list and serve reviews separately

* test(lens): cover list summaries, review polling and review access

* feat(lens): explain why a queued investigation is waiting

Works out whether no worker is connected, the worker is busy (with its running investigations and an estimated start time), or it is just being picked up.

* feat(lens): show the queue reason and what the worker is doing in the live strip

The progress header and the strip replace "Queued for your worker" with the concrete reason. While waiting, the strip lists the busy worker's investigations; click one to open it.

* feat(lens): add a review page model carrying the total reviewed count

* fix(lens): page live reviews by index so out-of-order reviews are never skipped

* feat(lens): take an index cursor on the reviews endpoint

* test(lens): cover index cursors across out-of-order and rolled-over reviews

* chore(ui): regenerate api types for the lens reviews endpoint

* feat(lens): page job reviews by index cursor

Adds api.reviews for GET /lens/{id}/runs/{job}/reviews?after=N, with a demo implementation. appendPage adds pages in arrival order and keeps the latest 200. liveJob now keys off reviewed, since the list no longer carries reviews.

* feat(ui): add a lens reviews query that polls the index cursor while live

* fix(lens): feed the live run from the reviews endpoint and keep View run open

LiveRun now gets its reviews from useJobReviews instead of the list, which no longer carries them. View run stays clickable while a run is queued or running, and before the first trace the opened view says what the worker is doing.

* fix(lens): split live conclusions into issues and patterns

A check could show up twice with the same label, once as an issue and once as a pattern.

* fix(lens): group live conclusions by check with short labels

There is now one group per check_id: issue traces are the main count and pattern traces a secondary note, so there are no duplicate red and grey cards. A long instruction falls back to the humanized check id. Adds briefReasoning and traceRows for the simplified trace list, and drops helpers nothing uses.

* feat(lens): simplify View run to traces and conclusions

The left pane is the trace list. A soft highlighter carrying the provider and model slides to the trace being reviewed, and clicking a row shows just Lens's reasoning and verdicts. The right pane keeps one conclusion card per check.

* refactor(lens): drop client-side replay in favour of real in-flight rows

Removes the playback reducer and its pacing. liveRows lists the traces the worker is reading, from job.reading, followed by completed reviews newest first, keyed by execution_id so a trace keeps its row when it finishes.

* feat(lens): show what the worker is reading and make View run obvious

Each trace in flight gets a highlighted row with the model and a live timer, and becomes its completed row in place. Completed rows show the real review time. View run is an outline button next to the progress line, and clicking anywhere on the strip opens it too.

* feat(lens): sum up a finished live run with time taken

doneLine reads like "Reviewed 30 traces in 31s with", measured from when reading started.

* feat(lens): slide one model rectangle over the traces being read

A single rounded rectangle carrying the provider logo and model wraps the real in-flight rows from job.reading. It translates and resizes over 250ms as traces finish in place. Before job.reading arrives it sits on a top slot showing the honest progress line, and when the run completes it fades out over 400ms. Rows have a fixed height and stable execution_id keys, so polls don't cause jumps or flicker.

* feat(lens): add in-flight runs to jobs and worker progress

* feat(lens): store in-flight runs from progress and clear them when a job ends

* refactor(lens): route progress, cancel and results through shared job transitions

* feat(lens): report each run as in flight when its review starts

* feat(lens): send in-flight runs with worker progress

* test(lens): cover in-flight runs across progress, old workers and terminal states

* test(lens): cover in-flight reporting under original run ids

* chore(ui): regenerate api types for lens in-flight runs

* feat(lens): model live reading lanes from in-flight runs and reviews

* feat(lens): show a now reading stage that types each trace's reasoning

* feat(lens): put the now reading stage above the trace list in View run

* fix(lens): resolve the analysis provider logo from the model catalog

* fix(lens): give demo jobs an empty in-flight list

* style(lens): format endpoint tests

* refactor(lens): name the run now handler in investigations view

* refactor(lens): name now reading conditions

* refactor(lens): name inline objects in the live run

* style(lens): format live run files

* fix(lens): keep worker settings inside the standalone worker package

* refactor(lens): keep update retry settings next to the repository

* fix(lens): start review history over when a run is reclaimed

* chore(lens): drop the unused review fixture

* refactor(lens): remove dead live helpers and use generated in-flight types

* fix(lens): keep polling a finished run until its last reviews arrive

* perf(lens): tick fast only while reasoning is typing

* fix(lens): isolate retried reviews and finding identities

* fix(lens): space the model name in run summary

* Update review.md

* fix(lens): make tool steps and conversations readable

* fix(lens): address trace rendering review and test failures

* fix(lens): preserve conversations with incomplete tool calls

* test(lens): retain failed tool styling coverage

* fix(lens): keep tool metadata in accessible result groups

* refactor(lens): build stable agent labels without mutation

* perf(lens): group and sort agent labels without repeated scans

* test(lens): await trace status filter option

---------

Co-authored-by: Ishaan Jaff <ishaan@berri.ai>
This commit is contained in:
moe-berri 2026-10-05 15:00:57 -07:00 • committed by GitHub
parent 7dc5b73f93
commit fe24be2e3d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
16 changed files with 614 additions and 146 deletions

View file

@ -24,6 +24,7 @@
"clsx": "^2.1.1",
"date-fns": "^4.4.0",
"dayjs": "1.11.19",
"es-toolkit": "1.49.0",
"jwt-decode": "4.0.0",
"lucide-react": "0.513.0",
"moment": "2.31.0",

View file

@ -41,6 +41,7 @@
"clsx": "^2.1.1",
"date-fns": "^4.4.0",
"dayjs": "1.11.19",
"es-toolkit": "1.49.0",
"jwt-decode": "4.0.0",
"lucide-react": "0.513.0",
"moment": "2.31.0",

View file

@ -6,7 +6,8 @@ import { useTracesApi } from "../../api";
import type { Span, SpanDetail, UIContent } from "../../types";
import { parseMessages, prettyPayload } from "../../utils";
import { Payload, TextBody } from "./PayloadBody";
import { payloadView } from "./payload";
import { ToolArguments } from "./ToolContent";
import { payloadView, toolInput } from "./payload";
import { Section } from "./Section";
import { ErrorBlock, StoredDiagnostic } from "./SpanError";
@ -41,7 +42,12 @@ function PayloadSection({
span: Span;
role: "input" | "output";
}) {
const view = payloadView(raw, content, span.type === "tool" && role === "output");
const view = payloadView(
raw,
content,
span.type === "tool" && role === "output",
span.type !== "tool" && role === "output",
);
const count = role === "input" ? messageCount(raw, content) : undefined;
return (
<Section
@ -50,13 +56,11 @@ function PayloadSection({
count={count}
defaultOpen={!count || count <= COLLAPSE_INPUT_ABOVE}
>
{(mode) =>
mode === "Raw" ? (
<TextBody text={prettyPayload(raw)} format="code" />
) : (
<Payload view={view} name={span.name} failed={span.status === "error"} />
)
}
{(mode) => {
if (mode === "Raw") return <TextBody text={prettyPayload(raw)} format="code" />;
if (span.type === "tool" && role === "input") return <ToolArguments args={toolInput(raw, content)} />;
return <Payload view={view} name={span.name} failed={span.status === "error"} />;
}}
</Section>
);
}

View file

@ -1,4 +1,4 @@
import { render, screen } from "@testing-library/react";
import { render, screen, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { describe, expect, it, vi } from "vitest";
@ -10,20 +10,22 @@ vi.mock("../../ui/spanProvider", () => ({ useSpanProvider: () => null }));
const LONG_QUERY = "Find every invoice for the customer that was billed twice. ".repeat(3).trim();
describe("ToolResultCard", () => {
it("expands a multiline result to its full text and keeps short results on one line", async () => {
it("shows multiline output immediately and expands long results without losing text", async () => {
const user = userEvent.setup();
const result = "line one\nline two\nTraceback: boom";
const { unmount } = render(<ToolResultCard name="read_file" result={result} failed />);
const toggle = screen.getByRole("button", { name: "Expand read_file result" });
expect(toggle).toHaveAttribute("aria-expanded", "false");
await user.click(toggle);
expect(screen.getByRole("button", { name: "Collapse read_file result" })).toHaveAttribute("aria-expanded", "true");
expect(screen.getByText((_, el) => el?.tagName === "PRE" && el.textContent === result)).toBeVisible();
unmount();
render(<ToolResultCard name="ls" result="No files found" />);
expect(screen.getByText("No files found")).toBeVisible();
expect(screen.queryByRole("button", { name: /^(Expand|Collapse) ls result$/ })).not.toBeInTheDocument();
const result = Array.from({ length: 30 }, (_, i) => `line ${i + 1}`).join("\n");
render(
<ToolResultCard name="terminal" result={JSON.stringify({ output: result, exit_code: 1, error: null })} failed />,
);
expect(screen.getByText(/line 1\s+line 2/, { selector: "pre" })).toBeVisible();
expect(screen.queryByText(/line 30/, { selector: "pre" })).not.toBeInTheDocument();
const failedResult = screen.getByRole("group", { name: "Failed tool result" });
expect(failedResult).toHaveClass("text-destructive");
expect(within(failedResult).getByText("exit_code")).toBeVisible();
expect(within(failedResult).getByText("1", { exact: true })).toBeVisible();
await user.click(screen.getByRole("button", { name: "Expand result" }));
expect(screen.getByText(/line 30/, { selector: "pre" })).toBeVisible();
await user.click(screen.getByRole("button", { name: "Collapse result" }));
expect(screen.queryByText(/line 30/, { selector: "pre" })).not.toBeInTheDocument();
});
});
@ -38,22 +40,17 @@ describe("MessageCard", () => {
expect(screen.getByText("Why was I billed twice?")).toBeVisible();
});
it("lists tool call arguments once and expands only the long value in place", async () => {
const user = userEvent.setup();
it("shows a tool's primary query immediately alongside its other arguments", () => {
render(
<MessageCard
message={{
role: "assistant",
content: "",
tool_calls: [{ name: "search_invoices", args: { customer_id: "acme-404", query: LONG_QUERY } }],
tool_calls: [{ name: "search_invoices", args: { customer_id: "test-404", query: LONG_QUERY } }],
}}
/>,
);
expect(screen.getByText("acme-404")).toBeInTheDocument();
expect(screen.queryByRole("button", { name: "Expand customer_id" })).not.toBeInTheDocument();
expect(screen.queryByText(LONG_QUERY, { selector: "pre" })).not.toBeInTheDocument();
await user.click(screen.getByRole("button", { name: "Expand query" }));
expect(screen.getByRole("button", { name: "Collapse query" })).toHaveAttribute("aria-expanded", "true");
expect(screen.getByText("test-404")).toBeVisible();
expect(screen.getByText(LONG_QUERY, { selector: "pre" })).toBeVisible();
});

View file

@ -1,17 +1,14 @@
"use client";
import { Bot, Settings2, User, Wrench } from "lucide-react";
import { useState } from "react";
import CopyButton from "@/components/shared/CopyButton";
import { cn } from "@/lib/cva.config";
import { FoldChevron } from "../../ui/Collapse";
import type { TraceMessage, TraceToolCall } from "../../types";
import { Block, BlockBadge, type BadgeTone } from "./Block";
import { FieldTree } from "./FieldTree";
import { Markdown } from "./Markdown";
import { fieldNode, type FieldEntry } from "./payload";
import { ToolArguments, ToolOutput } from "./ToolContent";
const ROLE_LABEL: Record<string, string> = { user: "User", system: "System", assistant: "Assistant", tool: "Tool" };
const ROLE_TONE: Record<string, BadgeTone> = { user: "user", assistant: "assistant", tool: "tool", system: "neutral" };
@ -24,12 +21,6 @@ const ROLE_ICON: Record<string, React.ReactNode> = {
const argsText = (args: unknown): string => (typeof args === "string" ? args : JSON.stringify(args));
function argEntries(args: unknown): readonly FieldEntry[] {
const node = fieldNode(args);
if (node.kind === "object" && node.entries.length > 0) return node.entries;
return [["arguments", node]];
}
export function ToolCallBlock({ call }: { call: TraceToolCall }) {
return (
<div className="rounded-md border border-border bg-background">
@ -46,8 +37,12 @@ export function ToolCallBlock({ call }: { call: TraceToolCall }) {
className="ml-auto size-6 text-muted-foreground"
/>
</div>
<div className="px-2.5 py-1">
<FieldTree entries={argEntries(call.args)} />
<div className="px-2.5 py-2">
{call.args === undefined ? (
<p className="py-2 text-xs text-muted-foreground">Arguments not recorded</p>
) : (
<ToolArguments args={call.args} />
)}
</div>
</div>
);
@ -90,41 +85,19 @@ export function MessageList({ messages }: { messages: readonly TraceMessage[] })
);
}
const LONG_RESULT_CHARS = 120;
/** A tool's result: one line by default, long or multiline results expand in place. Red when the tool failed. */
export function ToolResultCard({ name, result, failed = false }: { name: string; result: string; failed?: boolean }) {
const [open, setOpen] = useState(false);
const expandable = result.length > LONG_RESULT_CHARS || result.includes("\n");
const tone = failed ? "text-destructive" : "text-foreground";
return (
<article
className={cn(
"group/block overflow-hidden rounded-lg border bg-card",
"min-w-0 overflow-hidden rounded-lg border bg-card",
failed ? "border-destructive/40" : "border-border",
)}
>
<div className="flex min-h-9 items-center gap-2 py-1 pr-1.5 pl-2.5">
<div className="flex min-h-9 items-center gap-2 border-b border-border/70 px-3 py-1.5">
<BlockBadge tone={failed ? "failed" : "tool"}>
<Wrench />
</BlockBadge>
<span className={cn("shrink-0 font-mono text-xs font-medium", failed ? "text-destructive" : "text-foreground")}>
{name}
</span>
{expandable ? (
<button
type="button"
aria-expanded={open}
aria-label={`${open ? "Collapse" : "Expand"} ${name} result`}
onClick={() => setOpen((value) => !value)}
className={cn("flex min-w-0 flex-1 cursor-pointer items-center gap-1 text-left text-muted-foreground")}
>
<FoldChevron open={open} className="size-3.5 shrink-0" />
{!open && <span className="min-w-0 truncate text-sm">{result}</span>}
</button>
) : (
<span className={cn("min-w-0 flex-1 text-sm break-words", tone)}>{result || "No output"}</span>
)}
<span className="min-w-0 flex-1 truncate font-mono text-xs font-medium">{name}</span>
<CopyButton
variant="action"
value={result}
@ -133,16 +106,9 @@ export function ToolResultCard({ name, result, failed = false }: { name: string;
className="size-6 shrink-0 text-muted-foreground"
/>
</div>
{open && (
<pre
className={cn(
"max-h-96 overflow-auto border-t border-border/70 bg-muted/40 px-3 py-2.5 font-mono text-xs leading-relaxed break-words whitespace-pre-wrap",
tone,
)}
>
{result}
</pre>
)}
<div className="px-3 py-2.5">
<ToolOutput result={result} failed={failed} />
</div>
</article>
);
}

View file

@ -0,0 +1,77 @@
"use client";
import { useState } from "react";
import { cn } from "@/lib/cva.config";
import { FieldTree } from "./FieldTree";
import { MessageList } from "./Messages";
import { fieldNode, toolAction, toolResult, type FieldNode } from "./payload";
export function ToolText({ text, label }: { text: string; label: string }) {
const [expanded, setExpanded] = useState(false);
const preview = text.split("\n").slice(0, 8).join("\n").slice(0, 1200);
const truncated = preview.length < text.length;
return (
<div className="min-w-0">
<pre className="whitespace-pre-wrap break-words font-mono text-xs leading-5 wrap-anywhere">
{expanded ? text : preview || "No output recorded"}
{!expanded && truncated && "\n…"}
</pre>
{truncated && (
<button
type="button"
aria-expanded={expanded}
aria-label={`${expanded ? "Collapse" : "Expand"} ${label}`}
onClick={() => setExpanded((value) => !value)}
className="mt-2 rounded-sm text-xs font-medium text-muted-foreground underline-offset-4 hover:text-foreground hover:underline focus-visible:outline-2 focus-visible:outline-ring"
>
{expanded ? "Show less" : `Show full ${label} (${text.length.toLocaleString()} characters)`}
</button>
)}
</div>
);
}
function ToolValue({ node, label }: { node: FieldNode; label: string }) {
if (node.kind === "messages") return <MessageList messages={node.messages} />;
if (node.kind === "text" || node.kind === "scalar") return <ToolText text={node.text} label={label} />;
const entries = node.kind === "object" ? node.entries : node.items.map((item, i) => [String(i), item] as const);
return entries.length ? (
<FieldTree entries={entries} />
) : (
<ToolText text={node.kind === "object" ? "{}" : "[]"} label={label} />
);
}
export function ToolArguments({ args }: { args: unknown }) {
const node = fieldNode(args);
const action = toolAction(args);
const rest = node.kind === "object" ? node.entries.filter(([key]) => key !== action?.key) : [];
if (!action) return <ToolValue node={node} label="arguments" />;
return (
<div className="min-w-0 space-y-2">
<div className="rounded-md bg-muted/40 px-3 py-2.5">
<div className="mb-1.5 text-xs text-muted-foreground">{action.key}</div>
<ToolText text={action.text} label={action.key} />
</div>
{rest.length > 0 && <FieldTree entries={rest} />}
</div>
);
}
export function ToolOutput({ result, failed = false }: { result: string; failed?: boolean }) {
const { body, metadata } = toolResult(result);
return (
<div
role="group"
aria-label={failed ? "Failed tool result" : "Tool result"}
className={cn("min-w-0 space-y-3", failed && "text-destructive")}
>
<ToolValue node={body} label="result" />
{metadata.length > 0 && (
<div className="border-t border-border/70 pt-1">
<FieldTree entries={metadata} />
</div>
)}
</div>
);
}

View file

@ -1,6 +1,6 @@
import { describe, expect, it } from "vitest";
import { fieldNode, payloadView, textFormat } from "./payload";
import { fieldNode, payloadView, textFormat, toolInput, toolSummary, toolResult } from "./payload";
describe("textFormat", () => {
it.each([
@ -153,3 +153,63 @@ describe("payloadView", () => {
});
});
});
describe("tool payloads", () => {
it("decodes transport escapes once while preserving literal escapes inside a command", () => {
const args = { command: "printf 'first\\nsecond'\nls src", workdir: "/workspace" };
expect(toolInput(JSON.stringify(args))).toEqual(args);
expect(toolSummary(JSON.stringify(args))).toBe("printf 'first\\nsecond' ls src");
});
it("extracts a useful action from a truncated tree preview without showing JSON scaffolding", () => {
expect(toolSummary('{"command":"npm test\\n-- --run')).toBe("npm test -- --run");
expect(toolSummary('{"file_path":"/workspace/src/page.tsx","offset":10}')).toBe("/workspace/src/page.tsx");
expect(toolSummary('{"unknown":42}')).toBe("");
expect(toolSummary({ command: "[ -f package.json ] && npm test" })).toBe("[ -f package.json ] && npm test");
expect(toolSummary(JSON.stringify({ command: "{ npm test; }" }))).toBe("{ npm test; }");
});
it("preserves command failure metadata and renders structured results as fields", () => {
expect(toolResult(JSON.stringify({ output: "first\nsecond", exit_code: 1, error: null }))).toEqual({
body: { kind: "text", text: "first\nsecond", format: "plain" },
metadata: [
["exit_code", { kind: "scalar", text: "1" }],
["error", { kind: "scalar", text: "null" }],
],
});
expect(toolResult('{"result":{"number":42,"state":"open"}}').body).toEqual(
fieldNode({ result: { number: 42, state: "open" } }),
);
});
it("unwraps MCP text while preserving unknown blocks, annotations and error flags", () => {
expect(toolResult('{"content":[{"type":"text","text":"Permission denied"}],"isError":true}')).toEqual({
body: fieldNode("Permission denied"),
metadata: [["isError", { kind: "scalar", text: "true" }]],
});
const unknown = {
content: [
{ type: "image", data: "sample" },
{ type: "text", text: "caption", annotations: { audience: ["user"] } },
],
};
expect(toolResult(JSON.stringify(unknown)).body).toEqual(fieldNode(unknown));
expect(toolResult("{malformed output").body).toEqual(fieldNode("{malformed output"));
});
it("recognizes message arrays inside normalized text without rewriting ordinary agent content", () => {
const raw = JSON.stringify([{ role: "user", content: "SAVED TASK RESUMED: Continue this task" }]);
expect(payloadView(raw, { kind: "text", text: raw }, false)).toEqual({
kind: "messages",
messages: [{ role: "user", content: "SAVED TASK RESUMED: Continue this task" }],
});
const summary = JSON.stringify([{ content: "Checking files", tool_names: ["read_file"] }]);
expect(payloadView(summary, { kind: "text", text: summary }, false, true)).toMatchObject({
kind: "messages",
messages: [
{ role: "assistant", content: "Checking files", tool_calls: [{ name: "read_file", args: undefined }] },
],
});
expect(payloadView(summary, { kind: "text", text: summary }, false).kind).toBe("text");
});
});

View file

@ -1,5 +1,5 @@
import type { TraceMessage, UIContent, UIMessage } from "../../types";
import { parseJson, parseMessages } from "../../utils";
import { parseAssistantSummary, parseJson, parseMessages } from "../../utils";
export type TextFormat = "markdown" | "code" | "plain";
@ -63,6 +63,8 @@ const singleText = (messages: readonly UIMessage[]): string | null =>
messages.length === 1 && !messages[0].tool_calls?.length ? messages[0].content : null;
function textView(text: string): PayloadView {
const messages = parseMessages(text);
if (messages?.length) return { kind: "messages", messages };
return { kind: "text", text, format: textFormat(text) };
}
@ -93,6 +95,75 @@ function rawView(raw: string, toolOutput: boolean): PayloadView {
}
/** How a span's input or output reads best: the standard UI shape when the store sent one, else the raw payload. */
export function payloadView(raw: string, content: UIContent | undefined, toolOutput: boolean): PayloadView {
export function payloadView(
raw: string,
content: UIContent | undefined,
toolOutput: boolean,
assistantOutput = false,
): PayloadView {
if (assistantOutput && (!content || content.kind === "text")) {
const messages = parseAssistantSummary(content?.text ?? raw);
if (messages) return { kind: "messages", messages };
}
return content ? standardView(content, raw, toolOutput) : rawView(raw, toolOutput);
}
export function toolInput(raw: string, content?: UIContent): unknown {
const parsed = parseJson(raw);
if (parsed !== null) return parsed;
if (content?.kind === "fields") return Object.fromEntries(content.fields.map(({ key, value }) => [key, value]));
return content?.kind === "text" ? content.text : raw;
}
const ACTION_KEYS = ["command", "cmd", "code", "patch", "file_path", "path", "file", "query", "pattern", "url"];
export function toolAction(value: unknown): { key: string; text: string } | null {
if (typeof value === "string") {
const parsed = parseJson(value);
if (parsed !== null && typeof parsed === "object") return toolAction(parsed);
return value ? { key: "arguments", text: value } : null;
}
if (!value || typeof value !== "object" || Array.isArray(value)) return null;
for (const key of ACTION_KEYS) {
const text: unknown = Reflect.get(value, key);
if (typeof text === "string" && text) return { key, text };
}
return null;
}
export function toolSummary(raw: unknown): string {
const action = toolAction(raw);
if (action && (action.key !== "arguments" || !looksLikeJson(action.text))) return action.text.replace(/\s+/g, " ");
if (typeof raw !== "string") return "";
const match = /"(?:command|cmd|code|patch|file_path|path|file|query|pattern|url)"\s*:\s*"((?:[^"\\]|\\.)*)/.exec(raw);
if (!match) return "";
const text = match[1].replace(/\\u[0-9a-fA-F]{0,3}$|\\$/, "");
const decoded = parseJson(`"${text}"`);
return (typeof decoded === "string" ? decoded : text).replace(/\s+/g, " ");
}
export function toolResult(result: string): { body: FieldNode; metadata: readonly FieldEntry[] } {
const parsed = parseJson(result);
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
const entries = Object.entries(parsed);
const output = entries.find(([key, value]) => key === "output" && typeof value === "string");
if (output)
return {
body: fieldNode(output[1]),
metadata: entries.filter(([key]) => key !== "output").map(([key, value]) => [key, fieldNode(value)]),
};
const content: unknown = Reflect.get(parsed, "content");
const isTextBlock = (block: unknown): block is { type: "text"; text: string } => {
if (!block || typeof block !== "object") return false;
const hasText = Reflect.get(block, "type") === "text" && typeof Reflect.get(block, "text") === "string";
return hasText && Object.keys(block).every((key) => key === "type" || key === "text");
};
if (Array.isArray(content) && content.length && content.every(isTextBlock)) {
return {
body: fieldNode(content.map((block) => block.text).join("\n\n")),
metadata: entries.filter(([key]) => key !== "content").map(([key, value]) => [key, fieldNode(value)]),
};
}
}
return { body: fieldNode(typeof parsed === "string" ? parsed : result), metadata: [] };
}

View file

@ -56,7 +56,10 @@ describe("TraceConversation", () => {
expect(conversationTab).toHaveAttribute("aria-selected", "true");
const conversation = await screen.findByRole("region", { name: "Trace conversation" });
expect(await within(conversation).findByText("Read the release notes")).toBeVisible();
await user.click(await within(conversation).findByRole("button", { name: "Expand read_file tool call" }));
expect(await within(conversation).findByRole("button", { name: "Expand read_file tool call" })).toHaveTextContent(
"CHANGELOG.md",
);
await user.click(within(conversation).getByRole("button", { name: "Expand read_file tool call" }));
expect(await within(conversation).findByText("CHANGELOG.md")).toBeVisible();
expect(await within(conversation).findByText("All checks passed")).toBeVisible();
expect(await within(conversation).findByText("The release is ready")).toBeVisible();
@ -67,6 +70,56 @@ describe("TraceConversation", () => {
expect(screen.getByRole("heading", { name: "read_file" })).toBeVisible();
});
it("renders a failed shell exchange in both views and preserves its raw result", async () => {
const user = userEvent.setup();
const command = "npm test -- checkout\nprintf 'finished\\n'";
const output = JSON.stringify({ output: "PASS cart.test.ts\nFAIL checkout.test.ts", exit_code: 1, error: null });
const failedTool = { ...tool, name: "terminal", status: "error", error: null };
vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root, failedTool] } as Trace);
vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) =>
id === "root"
? rootDetail
: {
...toolDetail,
input: JSON.stringify({ command, workdir: "/workspace" }),
output,
input_ui: {
kind: "fields",
fields: [
{ key: "command", value: command },
{ key: "workdir", value: "/workspace" },
],
},
output_ui: {
kind: "fields",
fields: [
{ key: "output", value: "PASS cart.test.ts\nFAIL checkout.test.ts" },
{ key: "exit_code", value: "1" },
],
},
},
);
renderWithProviders(
<RoutedRunView traceId={trace.summary.trace_id} accessToken="test" onBack={vi.fn()} embedded />,
);
await user.click(await screen.findByRole("tab", { name: "Conversation" }));
const conversation = await screen.findByRole("region", { name: "Trace conversation" });
expect(await within(conversation).findByText(/npm test -- checkout/, { selector: "pre" })).toHaveTextContent(
"printf 'finished\\n'",
);
expect(within(conversation).getByText(/PASS cart.test.ts/, { selector: "pre" })).toHaveTextContent(
"FAIL checkout.test.ts",
);
expect(within(conversation).getByText("exit_code")).toBeVisible();
await user.click(within(conversation).getByRole("button", { name: "Inspect step terminal" }));
const details = screen.getByRole("complementary", { name: "Span details" });
expect(await within(details).findByText(/npm test -- checkout/, { selector: "pre" })).toBeVisible();
const result = within(details).getByRole("region", { name: "Output", exact: true });
expect(within(result).getByText("exit_code")).toBeVisible();
await user.click(within(result).getByRole("radio", { name: "Raw" }));
expect(within(result).getByText(/"exit_code": 1/, { selector: "pre" })).toHaveTextContent('"error": null');
});
it("loads only twenty full steps at a time and fetches the remainder on demand", async () => {
const user = userEvent.setup();
const spans = [
@ -85,6 +138,29 @@ describe("TraceConversation", () => {
expect(screen.queryByRole("button", { name: /Load next/ })).not.toBeInTheDocument();
});
it("shows distinct agent invocation labels together with each step's time", async () => {
const first = { ...root, name: "reviewer", start_offset_ms: 1000 };
const second = { ...first, span_id: "second", start_offset_ms: 2000 };
vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => ({
...rootDetail,
span_id: id,
input: id === "root" ? "Review code" : "Review tests",
output: "",
}));
renderWithProviders(
<TraceConversation
trace={{ ...trace, spans: [first, second] } as Trace}
accessToken="test"
onOpenStep={vi.fn()}
/>,
);
expect(await screen.findByText("reviewer (1)")).toBeVisible();
expect(screen.getByText("reviewer (2)")).toBeVisible();
const steps = screen.getAllByRole("region", { name: "Conversation step reviewer" });
expect(within(steps[0]).getByText("1.00s")).toBeVisible();
expect(within(steps[1]).getByText("2.00s")).toBeVisible();
});
it.each([rootDetail.output, ""])(
"keeps the root failure visible while loading, then places it in order (output: %s)",
async (output) => {

View file

@ -5,12 +5,14 @@ import { useState } from "react";
import { ChevronRight, Wrench } from "lucide-react";
import { cn } from "@/lib/cva.config";
import CopyButton from "@/components/shared/CopyButton";
import { ToolArguments, ToolOutput } from "../content/ToolContent";
import { toolSummary } from "../content/payload";
import { useTracesApi } from "../../api";
import { Button } from "@/components/ui/button";
import { buildConversation, conversationSteps, CONVERSATION_PAGE_SIZE, type ConversationItem } from "./conversation";
import { ErrorBlock } from "../content/SpanError";
import { Markdown } from "../content/Markdown";
import { ToolCallBlock } from "../content/Messages";
import { ToolCallBlock, ToolResultCard } from "../content/Messages";
import type { SpanDetail, Trace, TraceMessage } from "../../types";
import { fmtMs } from "../../utils";
@ -45,6 +47,7 @@ export function TraceConversation({
);
const complete = loadedCount === steps.length;
const items = buildConversation(trace.spans, details, complete);
const multipleAgents = new Set(items.map((item) => item.agentId).filter(Boolean)).size > 1;
const inlineErrorIds = new Set(items.filter((item) => item.showError).map((item) => item.span.span_id));
const rootErrors = trace.spans.filter((span) => {
const failedRoot = span.parent_span_id === null && span.status === "error" && span.type !== "tool";
@ -52,31 +55,40 @@ export function TraceConversation({
});
return (
<section aria-label="Trace conversation" className="min-h-0 flex-1 overflow-y-auto">
<div className="mx-auto max-w-3xl space-y-6 px-6 py-5">
<div className="mx-auto min-w-0 max-w-3xl space-y-5 px-3 py-4 sm:px-6 sm:py-5">
{rootErrors.map((span) => (
<ErrorBlock key={span.span_id} span={span} />
))}
{items.map((item) => (
<section
key={item.id}
className="group/conversation relative space-y-3"
aria-label={`Conversation step ${item.span.name}`}
>
<section key={item.id} className="min-w-0 space-y-2" aria-label={`Conversation step ${item.span.name}`}>
{(multipleAgents || item.toolResult === undefined) && (
<div className="flex min-w-0 items-center justify-between gap-2 text-xs text-muted-foreground">
<div className="flex min-w-0 items-center gap-2">
{multipleAgents && (
<span className="truncate" title={item.agentName}>
{item.agentName}
</span>
)}
<span className="shrink-0 tabular-nums">{fmtMs(item.span.start_offset_ms)}</span>
</div>
{item.toolResult === undefined && (
<Button
variant="ghost"
size="xs"
aria-label={`Inspect step ${item.span.name}`}
onClick={() => onOpenStep(item.span.span_id)}
className="shrink-0 text-muted-foreground"
>
Inspect step
</Button>
)}
</div>
)}
{item.showError && <ErrorBlock span={item.span} />}
{item.messages.map((message, index) => (
<ConversationMessage key={index} message={message} />
))}
{item.toolResult !== undefined && <ConversationTool item={item} />}
<Button
variant="ghost"
size="xs"
title={`${item.span.agent || item.span.name} · ${fmtMs(item.span.start_offset_ms)}`}
aria-label={`Inspect step ${item.span.name}`}
className="absolute right-0 -bottom-5 z-raised bg-background text-muted-foreground opacity-0 group-hover/conversation:opacity-100 focus-visible:opacity-100"
onClick={() => onOpenStep(item.span.span_id)}
>
Inspect step
</Button>
{item.toolResult !== undefined && <ConversationTool item={item} onOpenStep={onOpenStep} />}
</section>
))}
{queries.map(
@ -116,30 +128,49 @@ export function TraceConversation({
);
}
function ConversationTool({ item }: { item: ConversationItem }) {
const [open, setOpen] = useState(false);
function ConversationTool({ item, onOpenStep }: { item: ConversationItem; onOpenStep: (id: string) => void }) {
const [open, setOpen] = useState(item.span.status === "error");
const failed = item.span.status === "error";
const summary = toolSummary(item.toolCall?.args);
return (
<div className="rounded-md border">
<button
type="button"
aria-expanded={open}
aria-label={`${open ? "Collapse" : "Expand"} ${item.span.name} tool call`}
onClick={() => setOpen((value) => !value)}
className="flex w-full items-center gap-2 px-3 py-2.5 text-left text-sm hover:bg-muted/40"
>
<ChevronRight
className={cn("size-3.5 shrink-0 text-muted-foreground transition-transform", open && "rotate-90")}
/>
<Wrench className="size-3.5 shrink-0 text-muted-foreground" />
<span className="min-w-0 truncate font-medium">{item.span.name}</span>
<span className={cn("ml-auto shrink-0 text-xs", failed ? "text-destructive" : "text-muted-foreground")}>
{failed ? "Failed" : "Completed"}
</span>
</button>
<div className={cn("min-w-0 overflow-hidden rounded-md border", failed && "border-destructive/40")}>
<div className="flex min-w-0 items-start gap-1 pr-2">
<button
type="button"
aria-expanded={open}
aria-label={`${open ? "Collapse" : "Expand"} ${item.span.name} tool call`}
onClick={() => setOpen((value) => !value)}
className="flex min-w-0 flex-1 items-center gap-2 px-3 py-2.5 text-left text-sm hover:bg-muted/40"
>
<ChevronRight
className={cn("size-3.5 shrink-0 text-muted-foreground transition-transform", open && "rotate-90")}
/>
<Wrench className="size-3.5 shrink-0 text-muted-foreground" />
<span className="min-w-0 flex-1">
<span className="block truncate font-medium">{item.span.name}</span>
{summary && !open && (
<span className="mt-1 block truncate font-mono text-xs text-muted-foreground" title={summary}>
{summary}
</span>
)}
</span>
<span className={cn("ml-auto shrink-0 text-xs", failed ? "text-destructive" : "text-muted-foreground")}>
{failed ? "Failed" : "Completed"}
</span>
</button>
<Button
variant="ghost"
size="xs"
aria-label={`Inspect step ${item.span.name}`}
onClick={() => onOpenStep(item.span.span_id)}
className="my-2.5 shrink-0 text-muted-foreground"
>
Inspect
</Button>
</div>
{open && (
<div className="space-y-3 border-t px-3 py-3">
{item.toolCall && <ToolCallBlock call={item.toolCall} />}
<div className="min-w-0 space-y-3 border-t px-3 py-3">
{item.toolCall && <ToolArguments args={item.toolCall.args} />}
{item.span.error && item.span.error !== item.toolResult && (
<p className="text-sm text-destructive">{item.span.error}</p>
)}
@ -152,9 +183,7 @@ function ConversationTool({ item }: { item: ConversationItem }) {
iconOnly
/>
</div>
<pre className="max-h-80 overflow-auto whitespace-pre-wrap break-words text-xs leading-5">
{item.toolResult || "No output recorded"}
</pre>
<ToolOutput result={item.toolResult ?? ""} failed={failed} />
</div>
)}
</div>
@ -171,13 +200,7 @@ function ConversationMessage({ message }: { message: TraceMessage }) {
</div>
</details>
);
if (message.role === "tool")
return (
<details className="rounded-md border p-3 text-sm">
<summary className="cursor-pointer">{message.name || "Tool result"}</summary>
<pre className="max-h-80 overflow-auto whitespace-pre-wrap pt-3 text-xs">{message.content}</pre>
</details>
);
if (message.role === "tool") return <ToolResultCard name={message.name || "Tool result"} result={message.content} />;
return (
<div className={message.role === "user" ? "flex justify-end" : "space-y-3"}>
{message.content && (

View file

@ -20,6 +20,40 @@ const detail = (span_id: string, input: unknown, output: unknown): SpanDetail =>
});
describe("trace conversation", () => {
it.each(["reviewer", "__proto__", "constructor"])("keeps repeated agent name %s stable as steps load", (name) => {
const first = { ...root, span_id: "first", name, start_offset_ms: 1 };
const second = { ...first, span_id: "second", start_offset_ms: 2 };
const spans = [second, first];
const firstDetails = new Map([["first", detail("first", "Review code", "")]]);
const partial = buildConversation(spans, firstDetails, false);
const complete = buildConversation(
spans,
new Map([...firstDetails, ["second", detail("second", "Review tests", "")]]),
true,
);
expect(partial.map(({ agentId, agentName }) => ({ agentId, agentName }))).toEqual([
{ agentId: "first", agentName: `${name} (1)` },
]);
expect(complete.map(({ agentId, agentName }) => ({ agentId, agentName }))).toEqual([
{ agentId: "first", agentName: `${name} (1)` },
{ agentId: "second", agentName: `${name} (2)` },
]);
});
it("breaks equal-time agent label ties by span ID without changing event order", () => {
const first = { ...root, span_id: "first", name: "reviewer", start_offset_ms: 1 };
const second = { ...first, span_id: "second" };
const details = new Map([
["first", detail("first", "Review code", "")],
["second", detail("second", "Review tests", "")],
]);
const items = buildConversation([second, first], details, true);
expect(items.map(({ agentId, agentName }) => ({ agentId, agentName }))).toEqual([
{ agentId: "second", agentName: "reviewer (2)" },
{ agentId: "first", agentName: "reviewer (1)" },
]);
});
it("removes repeated prefixes and trimmed context, but preserves a genuinely repeated question", () => {
expect(newConversationMessages([user, call, result], [user, call, result, answer])).toEqual([answer]);
expect(newConversationMessages([user, call, result], [call, result, answer])).toEqual([answer]);
@ -252,3 +286,56 @@ describe("trace conversation", () => {
expect(items.flatMap((item) => item.messages)).toEqual([answer, answer]);
});
});
describe("recorded tool summaries", () => {
it("pairs repeated named calls within their own branch and preserves unmatched calls and custom text", () => {
const model = { ...root, span_id: "model", parent_span_id: "root", type: "llm", start_offset_ms: 1 } as Span;
const first = { ...model, span_id: "first", name: "terminal", type: "tool", start_offset_ms: 2 } as Span;
const second = { ...first, span_id: "second", start_offset_ms: 3 };
const child = { ...root, span_id: "child", parent_span_id: "root", start_offset_ms: 4 };
const childTool = { ...first, span_id: "child-tool", parent_span_id: "child", start_offset_ms: 5 };
const saved = "SAVED TASK RESUMED: Continue the unfinished task exactly as recorded";
const summary = JSON.stringify([{ content: "Checking", tool_names: ["terminal", "terminal", "read_file"] }]);
const details = new Map([
["root", detail("root", [{ role: "user", content: saved }], [])],
[
"model",
{ ...detail("model", [], []), output: summary, output_ui: { kind: "text", text: summary } } as SpanDetail,
],
["first", detail("first", { command: "pwd" }, { output: "/workspace", exit_code: 0 })],
["second", detail("second", { command: "pwd" }, { output: "/workspace", exit_code: 0 })],
["child", detail("child", [{ role: "user", content: saved }], [])],
["child-tool", detail("child-tool", { path: "README.md" }, "content")],
]);
const items = buildConversation([root, model, first, second, child, childTool], details, true);
expect(items.flatMap((item) => item.messages).filter((message) => message.content === saved)).toHaveLength(2);
expect(items.flatMap((item) => item.messages.flatMap((message) => message.tool_calls ?? []))).toEqual([
{ name: "read_file", args: undefined },
]);
expect(items.filter((item) => item.toolCall).map((item) => item.id)).toEqual(["first", "second", "child-tool"]);
expect(JSON.parse(items.find((item) => item.id === "first")!.toolResult!)).toEqual({
output: "/workspace",
exit_code: 0,
});
});
it("deduplicates native OpenAI function calls without requiring message text", () => {
const model = { ...root, span_id: "model", parent_span_id: "root", type: "llm", start_offset_ms: 1 } as Span;
const tool = { ...model, span_id: "tool", name: "read_file", type: "tool", start_offset_ms: 2 } as Span;
const output = [
{
role: "assistant",
content: null,
tool_calls: [{ type: "function", function: { name: "read_file", arguments: '{"path":"README.md"}' } }],
},
];
const details = new Map([
["root", detail("root", [], [])],
["model", detail("model", [], output)],
["tool", detail("tool", { path: "README.md" }, "file content")],
]);
expect(
buildConversation([root, model, tool], details, true).filter((item) => item.toolCall || item.messages.length),
).toHaveLength(1);
});
});

View file

@ -1,6 +1,6 @@
import type { Span, SpanDetail, TraceMessage, TraceToolCall, UIContent } from "../../types";
import { isFrameworkSpan, parseJson, parseMessages, prettyPayload } from "../../utils";
import { toTraceMessage } from "../content/payload";
import { isFrameworkSpan, parseAssistantSummary, parseMessages, prettyPayload } from "../../utils";
import { toTraceMessage, toolInput } from "../content/payload";
export const CONVERSATION_PAGE_SIZE = 20;
@ -24,7 +24,8 @@ function contentText(value: string, content?: UIContent): string {
function messages(value: string, content: UIContent | undefined, role: string): TraceMessage[] {
if (content?.kind === "messages") return content.messages.map(toTraceMessage);
const parsed = !content ? parseMessages(value) : null;
const source = content?.kind === "text" ? content.text : value;
const parsed = parseMessages(source) ?? (role === "assistant" ? parseAssistantSummary(source) : null);
if (parsed) return parsed;
const text = contentText(value, content);
return text ? [{ role, content: text }] : [];
@ -69,6 +70,8 @@ export interface ConversationItem {
messages: TraceMessage[];
toolCall?: TraceToolCall;
toolResult?: string;
agentId?: string;
agentName?: string;
showError?: boolean;
}
@ -78,16 +81,13 @@ function toolItem(
pending: TraceToolCall[],
items: ConversationItem[],
): ConversationItem {
const args =
parseJson(detail.input) ??
(detail.input_ui?.kind === "fields"
? Object.fromEntries(detail.input_ui.fields.map((field) => [field.key, field.value]))
: detail.input);
const args = toolInput(detail.input, detail.input_ui);
const call = { name: span.name, args };
const match = pending.findIndex(
(candidate) =>
candidate.name === call.name &&
JSON.stringify(stableValue(candidate.args)) === JSON.stringify(stableValue(call.args)),
(candidate.args === undefined ||
JSON.stringify(stableValue(candidate.args)) === JSON.stringify(stableValue(call.args))),
);
if (match >= 0) {
const [matched] = pending.splice(match, 1);
@ -97,7 +97,10 @@ function toolItem(
message.tool_calls = message.tool_calls.filter((call) => call !== matched);
}
}
const result = contentText(detail.output, detail.output_ui);
const result =
detail.output_ui?.kind === "text"
? detail.output_ui.text
: prettyPayload(detail.output) || contentText(detail.output, detail.output_ui);
return { id: span.span_id, span, messages: [], toolCall: call, toolResult: result };
}
@ -168,6 +171,17 @@ function withoutForwardedAnswers(
);
}
function agentLabels(agents: readonly Span[]): ReadonlyMap<string, string> {
const named = agents.map((agent) => ({ agent, name: agent.name || agent.agent || "Agent" }));
return new Map(
Object.values(groupBy(named, ({ name }) => JSON.stringify(name))).flatMap((group) =>
orderBy(group, [({ agent }) => agent.start_offset_ms, ({ agent }) => agent.span_id], ["asc", "asc"]).map(
({ agent, name }, index) => [agent.span_id, group.length > 1 ? `${name} (${index + 1})` : name] as const,
),
),
);
}
export function buildConversation(
spans: readonly Span[],
details: ReadonlyMap<string, SpanDetail>,
@ -241,10 +255,18 @@ export function buildConversation(
};
if (combined.length || item.showError) items.push(item);
}
const agents = [...new Set(conversationSteps(spans).map(branch))].flatMap((id) => {
const span = byId.get(id);
return span ? [span] : [];
});
const labels = agentLabels(agents);
return items
.map((item) => ({
...item,
agentId: branch(item.span),
agentName: labels.get(branch(item.span)) || item.span.agent,
messages: item.messages.filter((message) => Boolean(message.content) || Boolean(message.tool_calls?.length)),
}))
.filter((item) => item.messages.length || item.toolResult !== undefined || item.showError);
}
import { groupBy, orderBy } from "es-toolkit";

View file

@ -341,12 +341,14 @@ describe("DetailPane", () => {
expect(output).not.toHaveTextContent("raw output left unparsed");
});
it("keeps the failed-tool styling when a tool's output arrives as a single message", async () => {
it("identifies and styles a failed tool result when its output arrives as a single message", async () => {
vi.mocked(agentTraceSpanCall).mockResolvedValue(failedToolMessageDetail);
renderPane(spanRow(failedTool));
const output = await screen.findByRole("region", { name: "Output" });
const result = within(output).getByText("permission denied: /etc/shadow");
const result = within(output).getByRole("group", { name: "Failed tool result" });
expect(result).toBeVisible();
expect(result).toHaveClass("text-destructive");
expect(result).toHaveTextContent("permission denied: /etc/shadow");
expect(output).not.toHaveTextContent("Assistant");
});

View file

@ -10,6 +10,7 @@ import type { GroupRowData, LoadMoreRowData, SpanRowData } from "../../tree";
import type { SpanType } from "../../types";
import { fmtMs, previewText, type TreeGuide } from "../../utils";
import { groupFacts, SpanHoverCard, spanFacts } from "./SpanHoverCard";
import { toolSummary } from "../content/payload";
import { BAR_TRACK, barGeometry } from "./timeline";
export type TreeLayout = "tree" | "waterfall";
@ -146,6 +147,7 @@ const readablePreview = (preview: string): string => {
const subtitle = (row: SpanRowData, filtering: boolean): string => {
const { span } = row;
if (span.type === "tool") return toolSummary(span.input_preview);
if (filtering) return readablePreview(span.input_preview) || span.agent;
return span.type === "agent" && span.parent_span_id ? readablePreview(span.input_preview) : "";
};

View file

@ -300,10 +300,42 @@ describe("payload helpers", () => {
expect(
parseMessages(JSON.stringify({ role: "assistant", content: [{ type: "text", text: "An execution record" }] })),
).toEqual([{ role: "assistant", content: "An execution record" }]);
expect(parseMessages(JSON.stringify({ role: "assistant", content: "hello", tool_calls: {} }))).toEqual([
{ role: "assistant", content: "hello", tool_calls: [{ name: "Tool call", args: {} }] },
]);
expect(
parseMessages(
JSON.stringify({ role: "assistant", content: "hello", tool_calls: [{ name: "lookup", args: "42" }] }),
),
).toEqual([{ role: "assistant", content: "hello", tool_calls: [{ name: "lookup", args: "42" }] }]);
expect(parseMessages('[{"role":"assistant","tool_calls":[]}]')).toBeNull();
expect(parseMessages('[{"role":"user","content":42}]')).toBeNull();
});
it("preserves conversation text and incomplete calls beside valid function calls", () => {
const incomplete = { id: "pending", function: { arguments: '{"path":"README.md"}' } };
const tool_calls = [incomplete, { function: { name: "read_file", arguments: '{"path":"AGENTS.md"}' } }, null];
expect(
parseMessages(
JSON.stringify([
{ role: "user", content: "Read the project instructions" },
{ role: "assistant", content: "Checking the files", tool_calls },
]),
),
).toEqual([
{ role: "user", content: "Read the project instructions" },
{
role: "assistant",
content: "Checking the files",
tool_calls: [
{ name: "Tool call", args: incomplete },
{ name: "read_file", args: { path: "AGENTS.md" } },
{ name: "Tool call", args: null },
],
},
]);
});
it("reads LangChain's serialized messages with their roles, names and tool calls", () => {
const dumped = [
{ type: "human", data: { content: "What is an agent trace?", name: null } },

View file

@ -323,6 +323,9 @@ const LANGCHAIN_ROLE: Readonly<Record<string, string>> = {
tool: "tool",
};
const isMessageContent = (value: unknown): value is string | unknown[] =>
typeof value === "string" || Array.isArray(value);
const langchainToolCalls = (data: object): TraceToolCall[] | undefined => {
const calls: unknown = Reflect.get(data, "tool_calls");
if (!Array.isArray(calls) || calls.length === 0) return undefined;
@ -337,7 +340,7 @@ const parseLangchainMessage = (value: object): TraceMessage | null => {
const data: unknown = Reflect.get(value, "data");
if (!role || typeof data !== "object" || data === null) return null;
const content: unknown = Reflect.get(data, "content");
if (typeof content !== "string" && !Array.isArray(content)) return null;
if (!isMessageContent(content)) return null;
const name: unknown = Reflect.get(data, "name");
const toolCalls = langchainToolCalls(data);
return {
@ -348,15 +351,59 @@ const parseLangchainMessage = (value: object): TraceMessage | null => {
};
};
const parseToolCall = (call: unknown): TraceToolCall => {
const unknownCall = { name: "Tool call", args: call };
if (!call || typeof call !== "object") return unknownCall;
const fn: unknown = Reflect.get(call, "function");
const source = fn && typeof fn === "object" ? fn : call;
const name: unknown = Reflect.get(source, "name");
if (typeof name !== "string" || !name) return unknownCall;
if ("args" in source) return { name, args: source.args };
const args: unknown = Reflect.get(source, "arguments");
return { name, args: typeof args === "string" ? parseJson(args) ?? args : args };
};
const parseMessage = (value: unknown): TraceMessage | null => {
if (typeof value !== "object" || value === null) return null;
if (!("role" in value) && "data" in value) return parseLangchainMessage(value);
const role: unknown = Reflect.get(value, "role");
const content: unknown = Reflect.get(value, "content") ?? Reflect.get(value, "parts");
if (typeof role !== "string" || (typeof content !== "string" && !Array.isArray(content))) return null;
return { ...value, role, content: messageText(typeof content === "string" ? content : JSON.stringify(content)) };
const rawCalls: unknown = Reflect.get(value, "tool_calls");
const callEntries = Array.isArray(rawCalls) ? rawCalls : [rawCalls];
const hasCalls = rawCalls != null && callEntries.length > 0;
const emptyToolMessage = content == null && hasCalls;
const hasContent = isMessageContent(content) || emptyToolMessage;
if (typeof role !== "string" || !hasContent) return null;
const calls = callEntries.map(parseToolCall);
const text = typeof content === "string" ? content : JSON.stringify(content ?? "");
return {
...value,
role,
content: content == null ? "" : messageText(text),
...(rawCalls != null ? { tool_calls: calls } : {}),
};
};
export function parseAssistantSummary(value: string): TraceMessage[] | null {
const parsed = parseJson(value);
if (!Array.isArray(parsed) || !parsed.length) return null;
const isSummary = (item: unknown): item is { content: string | null; tool_names: string[] } => {
if (!item || typeof item !== "object") return false;
const content: unknown = Reflect.get(item, "content");
const tools: unknown = Reflect.get(item, "tool_names");
const knownFields = Object.keys(item).every((key) => key === "content" || key === "tool_names");
const validContent = content === null || typeof content === "string";
const validTools = Array.isArray(tools) && tools.every((name: unknown) => typeof name === "string");
return knownFields && validContent && validTools;
};
if (!parsed.every(isSummary)) return null;
return parsed.map((item) => ({
role: "assistant",
content: item.content ?? "",
tool_calls: item.tool_names.map((name: string) => ({ name, args: undefined })),
}));
}
/** An llm span's input (array of messages) or output (one message); null when it isn't one. */
export function parseMessages(value: string): TraceMessage[] | null {
const parsed = parseJson(value);