feat(lens): read review spans as a conversation timeline

Turns spans into the user's ask, tool calls with args and results, and the agent's reply, dropping system prompts. Also handles a preview cut that lands inside the Output header.
This commit is contained in:
Ishaan Jaff 2026-10-03 17:22:13 -07:00
parent b3b7159393
commit 7de044b71e
No known key found for this signature in database
2 changed files with 261 additions and 90 deletions

View file

@ -1,51 +1,126 @@
import { describe, expect, it } from "vitest";
import { spanPreviewLines } from "./spanPreview";
import { assistantReply, stepFailure, timeline, toolCall, userAsk } from "./spanPreview";
describe("span previews", () => {
it("shows the last chat message of an agent span as role: content, not raw JSON", () => {
const preview =
'Input: [{"role": "system", "content": "You are a research assistant."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find it."}]';
expect(spanPreviewLines(preview)).toEqual([
{ label: "in", text: "user: Find the median latency of the EU region.", error: false },
{ label: "out", text: "assistant: I could not find it.", error: false },
]);
const OMIT = "\n[... preview omitted; read this span for evidence ...]\n";
const span = (kind: string, name: string, preview: string, cited = false) => ({
span_id: `${kind}-${name}`,
kind,
name,
preview,
cited,
});
const agentPreview =
'Input: [{"role": "system", "content": "You are a research assistant. Cite sources."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find a source for that."}]';
describe("user ask", () => {
it("takes the user's message and never the system prompt", () => {
expect(userAsk(agentPreview)).toBe("Find the median latency of the EU region.");
});
it("recovers the last message from truncated JSON with the omitted marker", () => {
const preview =
'Input: [{"role": "system", "content": "You are a bill\n[... preview omitted; read this span for evidence ...]\nn"}, {"role": "user", "content": "Cancel C2473\'s subscript';
expect(spanPreviewLines(preview)).toEqual([
{ label: "in", text: "user: Cancel C2473's subscript", error: false },
]);
it("marks a user message cut off by the preview limit", () => {
const cut = 'Input: [{"role": "system", "content": "You are a billing agent."}, {"role": "user", "content": "Cancel C2473\'s subscript';
expect(userAsk(cut)).toBe("Cancel C2473's subscript…");
});
it("keeps tool payloads compact and drops an unset status", () => {
const preview = 'Input: {"query":"latency"}\nOutput: {"results": []}\nStatus: STATUS_CODE_UNSET ';
expect(spanPreviewLines(preview)).toEqual([
{ label: "in", text: '{"query":"latency"}', error: false },
{ label: "out", text: '{"results":[]}', error: false },
]);
});
it("marks a failed span's output and shows its error", () => {
const preview =
'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500 search index unavailable';
const lines = spanPreviewLines(preview);
expect(lines.at(-1)).toEqual({ label: "error", text: "500 search index unavailable", error: true });
expect(lines.find((l) => l.label === "out")?.error).toBe(true);
});
it("keeps multi-line payloads inside their section and collapses whitespace", () => {
const preview = 'Output: {"content": "def total(items):\\n return 1"}\nmore text\nStatus: STATUS_CODE_OK';
expect(spanPreviewLines(preview)).toEqual([
{ label: "out", text: '{"content": "def total(items):\\n return 1"} more text', error: false },
]);
});
it("passes through plain text without sections", () => {
expect(spanPreviewLines("500 Internal Server Error")).toEqual([
{ label: "", text: "500 Internal Server Error", error: false },
]);
expect(spanPreviewLines("")).toEqual([]);
it("is null when only a system prompt is visible", () => {
expect(userAsk(`Input: [{"role": "system", "content": "You are a bill${OMIT}x"}]`)).toBeNull();
});
});
describe("assistant reply", () => {
it("reads the assistant output of an agent span", () => {
expect(assistantReply(agentPreview)).toBe("I could not find a source for that.");
});
it("recovers the end of a reply from a truncated llm span without its input", () => {
const llm = `Input: [{"role": "system", "content": "You are Acme's${OMIT}you want, I can also help you draft the message to your bank."}]\nStatus: STATUS_CODE_UNSET `;
expect(assistantReply(llm)).toBe("…you want, I can also help you draft the message to your bank.");
});
it("ignores a truncated tail that is a tool call or a JSON payload", () => {
const toolTail = `Input: [{"role": "system", "content": "You are a rese${OMIT}rch_docs", "arguments": "{\\"query\\":\\"latency\\"}"}}]}]\nStatus: STATUS_CODE_UNSET`;
expect(assistantReply(toolTail)).toBeNull();
});
it("is null when the output is empty", () => {
expect(assistantReply('Input: [{"role": "user", "content": "hi"}]\nOutput: \nStatus: STATUS_CODE_ERROR x')).toBeNull();
});
});
describe("tool call", () => {
it("names the tool and keeps args and result compact", () => {
const tool = span(
"tool",
"execute_tool lookup_order",
'Input: {"order_id":"A4160"}\nOutput: {"order_id": "A4160", "status": "processing"}\nStatus: STATUS_CODE_UNSET ',
);
expect(toolCall(tool)).toEqual({
kind: "tool",
name: "lookup_order",
args: '{"order_id":"A4160"}',
result: '{"order_id":"A4160","status":"processing"}',
error: false,
});
});
it("flags error status and error payloads but not ordinary results", () => {
const failed = 'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500';
expect(toolCall(span("tool", "execute_tool search_docs", failed)).error).toBe(true);
const denied = 'Input: {}\nOutput: {"status": "denied", "reason": "needs approval"}\nStatus: STATUS_CODE_UNSET';
expect(toolCall(span("tool", "execute_tool issue_refund", denied)).error).toBe(true);
const tests = 'Input: {}\nOutput: {"passed": 41, "failed": 3}\nStatus: STATUS_CODE_UNSET';
expect(toolCall(span("tool", "execute_tool run_tests", tests)).error).toBe(false);
});
it("reads a tool result that only survives in the preview tail", () => {
const cut = `Input: {"url":"https://example.com"}\nOutput: {"url": ${OMIT}, "text": "the figure is 37%"}\nStatus: STATUS_CODE_UNSET`;
expect(toolCall(span("tool", "execute_tool fetch_url", cut)).result).toBe('…, "text": "the figure is 37%"}');
});
it("keeps args clean when the preview cut lands inside the Output header", () => {
const cut = `Input: {"customer_id":"C3847"}\nO${OMIT}t month", "amount_usd": 416.67}\nStatus: STATUS_CODE_UNSET `;
const call = toolCall(span("tool", "execute_tool get_invoice", cut));
expect(call.args).toBe('{"customer_id":"C3847"}');
expect(call.result).toBe('…t month", "amount_usd": 416.67}');
});
});
describe("step failure", () => {
it("reports only non-ok statuses", () => {
expect(stepFailure("Output: \nStatus: STATUS_CODE_ERROR Request timed out.")).toBe("Request timed out.");
expect(stepFailure("Output: x\nStatus: STATUS_CODE_UNSET ")).toBeNull();
expect(stepFailure("plain")).toBeNull();
});
});
describe("timeline", () => {
it("reads like a conversation: ask, tool steps, then the final reply, without system text", () => {
const items = timeline([
span("agent", "invoke_agent research-agent", agentPreview),
span("llm", "chat gpt", `Input: [{"role": "system", "content": "You are a rese${OMIT}I could not find a source for that."}]`),
span(
"tool",
"execute_tool search_docs",
'Input: {"query":"EU latency"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500',
true,
),
]);
expect(items.map((i) => i.kind)).toEqual(["ask", "tool", "reply"]);
expect(items[0]).toMatchObject({ text: "Find the median latency of the EU region.", span: 0 });
expect(items[1]).toMatchObject({ name: "search_docs", error: true, span: 2 });
expect(items[2]).toMatchObject({ text: "I could not find a source for that.", span: 0 });
expect(JSON.stringify(items)).not.toContain("research assistant");
});
it("shows a failed model call as a failure step", () => {
const items = timeline([span("llm", "chat", "Input: x\nOutput: \nStatus: STATUS_CODE_ERROR Request timed out.")]);
expect(items).toEqual([{ kind: "failure", text: "Model call failed: Request timed out.", span: 0 }]);
});
it("falls back to short notes when nothing conversational can be recovered", () => {
const items = timeline([span("chain", "plan", 'Input: {"step": 1}')]);
expect(items).toEqual([{ kind: "note", label: "plan", text: '{"step":1}', span: 0 }]);
});
});

View file

@ -1,65 +1,161 @@
import { parseJson, parseMessages, previewText } from "@/components/view_logs/TraceView/traceUtils";
import { parseJson } from "@/components/view_logs/TraceView/traceUtils";
export interface PreviewLine {
label: string;
text: string;
error: boolean;
import type { Review } from "./types";
type Span = Review["spans"][number];
export type TimelineItem =
| { kind: "ask"; text: string; span: number }
| { kind: "tool"; name: string; args: string; result: string; error: boolean; span: number }
| { kind: "reply"; text: string; span: number }
| { kind: "failure"; text: string; span: number }
| { kind: "note"; label: string; text: string; span: number };
interface Message {
role: string;
content: string;
}
const SECTION = /^(Input|Output|Status): ?/;
const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/g;
const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/;
const CUT_OUTPUT_HEADER = /\nO(?:u(?:t(?:p(?:u(?:t:?)?)?)?)?)? ?$/;
const SECTION_START =/(?:^|\n)(Input|Output|Status): ?/g;
const OK_STATUS = /^STATUS_CODE_(UNSET|OK)\b/;
const ROLE_CONTENT = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)/g;
function sections(preview: string): Readonly<Record<string, string>> {
const lines = preview.split("\n");
const starts = lines.flatMap((line, index) => (SECTION.test(line) ? [index] : []));
if (!starts.length) return { Body: preview };
return Object.fromEntries(
starts.map((start, n) => {
const body = lines.slice(start, starts[n + 1] ?? lines.length).join("\n");
const name = SECTION.exec(body)?.[1] ?? "Body";
return [name, body.replace(SECTION, "").trim()];
}),
);
}
const MESSAGE = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)(")?/g;
const TOOL_PROBLEM = /"error"|\bdenied\b|\bforbidden\b|\bunauthori[sz]ed\b|\bnot (?:allowed|permitted)\b/i;
const SNIPPET = 160;
function decode(escaped: string): string {
const parsed = parseJson(`"${escaped.replace(/\\u[0-9a-fA-F]{0,3}$|\\$/, "")}"`);
return typeof parsed === "string" ? parsed : escaped;
}
function lastMessage(value: string): string | null {
const messages = parseMessages(value);
if (messages?.length) {
const last = messages.at(-1);
return last ? `${last.role}: ${last.content}` : null;
}
const matches = [...value.matchAll(ROLE_CONTENT)];
const match = matches.at(-1);
return match ? `${match[1]}: ${decode(match[2])}` : null;
function tidy(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
function compactJson(value: string): string {
const parsed = parseJson(value);
if (parsed === null || typeof parsed !== "object") return value;
return JSON.stringify(parsed);
function sections(text: string): Readonly<Partial<Record<"Input" | "Output" | "Status", string>>> {
const starts = [...text.matchAll(SECTION_START)];
return Object.fromEntries(
starts.map((match, n) => {
const from = (match.index ?? 0) + match[0].length;
const to = starts[n + 1]?.index ?? text.length;
return [match[1], text.slice(from, to)];
}),
);
}
function readable(value: string): string {
const joined = value.replace(OMITTED, " … ");
return (lastMessage(joined) ?? previewText(compactJson(joined))).replace(/\s+/g, " ").trim();
function parts(preview: string) {
const [rawHead, tail = ""] = preview.split(OMITTED);
const head = rawHead.replace(CUT_OUTPUT_HEADER, "\nOutput: ");
const whole = sections(preview.replace(OMITTED, "\n"));
const ending = sections(tail);
const before = sections(head);
const cutOutput = before.Output !== undefined ? `…${tail.split(/\nStatus: /)[0] ?? ""}` : "";
return {
input: before.Input ?? "",
output: tail ? ending.Output ?? cutOutput : whole.Output ?? "",
status: (whole.Status ?? "").trim(),
tail: tail.split(/\nStatus: /)[0] ?? "",
truncated: !!tail,
};
}
export function spanPreviewLines(preview: string): PreviewLine[] {
const parts = sections(preview);
const status = parts.Status ?? "";
function messages(text: string): Message[] {
return [...text.matchAll(MESSAGE)].map(([, role, content, closed]) => ({
role,
content: tidy(decode(content)) + (closed ? "" : "…"),
}));
}
function compact(text: string): string {
const trimmed = text.trim();
const parsed = parseJson(trimmed);
const flat = parsed !== null && typeof parsed === "object" ? JSON.stringify(parsed) : trimmed;
return tidy(flat).slice(0, SNIPPET);
}
export function stepFailure(preview: string): string | null {
const { status } = parts(preview);
if (!status || OK_STATUS.test(status)) return null;
return status.replace(/^STATUS_CODE_ERROR\s*/, "") || "failed";
}
export function userAsk(preview: string): string | null {
const asked = messages(parts(preview).input).filter((m) => m.role === "user" || m.role === "human");
return asked.at(-1)?.content || null;
}
function replyFromTail(tail: string): string | null {
const end = tail.trimEnd();
if (!end.endsWith('"}]') || end.includes('\\"') || /"function"|\{"|":\s/.test(end)) return null;
const text = tidy(end.slice(0, -3));
return text ? `…${text}` : null;
}
export function assistantReply(preview: string): string | null {
const { output, tail, truncated } = parts(preview);
const said = messages(output).filter((m) => m.role === "assistant" && m.content && m.content !== "…");
if (said.length) return said.at(-1)?.content ?? null;
return truncated && !output ? replyFromTail(tail) : null;
}
export function toolCall(span: Pick<Span, "name" | "preview">): Omit<Extract<TimelineItem, { kind: "tool" }>, "span"> {
const { input, output, status } = parts(span.preview);
const result = compact(output);
const failed = !!status && !OK_STATUS.test(status);
const content = [
...(parts.Body ? [{ label: "", text: readable(parts.Body), error: false }] : []),
...(parts.Input ? [{ label: "in", text: readable(parts.Input), error: false }] : []),
...(parts.Output ? [{ label: "out", text: readable(parts.Output), error: failed }] : []),
...(failed ? [{ label: "error", text: status.replace(/^STATUS_CODE_ERROR\s*/, ""), error: true }] : []),
];
return content.filter((line) => line.text);
return {
kind: "tool",
name: span.name.replace(/^execute_tool\s+/, ""),
args: compact(input),
result: result || (failed ? stepFailure(span.preview) ?? "" : ""),
error: failed || TOOL_PROBLEM.test(output),
};
}
function overlaps(a: string, b: string): boolean {
const strip = (s: string) => s.replace(/…/g, "").trim();
const [x, y] = [strip(a), strip(b)];
return !!x && !!y && (x.includes(y) || y.includes(x));
}
function itemFor(span: Span, index: number): TimelineItem | null {
if (span.kind === "tool") return { ...toolCall(span), span: index };
if (span.kind === "llm") {
const failure = stepFailure(span.preview);
if (failure) return { kind: "failure", text: `Model call failed: ${failure}`, span: index };
const reply = assistantReply(span.preview);
return reply ? { kind: "reply", text: reply, span: index } : null;
}
return null;
}
function fallback(spans: readonly Span[]): TimelineItem[] {
return spans.map((span, index) => {
const { input, output } = parts(span.preview);
const said = messages(input).filter((m) => m.role !== "system").at(-1)?.content;
return { kind: "note", label: span.name, text: said ?? compact(output || input), span: index };
});
}
export function timeline(spans: readonly Span[]): TimelineItem[] {
const askAt = spans.findIndex((span) => userAsk(span.preview));
const ask = askAt >= 0 ? userAsk(spans[askAt].preview) : null;
const agentAt = spans.findIndex((span) => span.kind === "agent" && assistantReply(span.preview));
const final = agentAt >= 0 ? assistantReply(spans[agentAt].preview) : null;
const steps = spans.flatMap((span, index) => itemFor(span, index) ?? []);
const middle = final ? steps.filter((item) => item.kind !== "reply" || !overlaps(item.text, final)) : steps;
const items: TimelineItem[] = [
...(ask ? [{ kind: "ask" as const, text: ask, span: askAt }] : []),
...middle,
...(final ? [{ kind: "reply" as const, text: final, span: agentAt }] : []),
];
return items.length ? items : fallback(spans);
}
export function spanPreviewLines(preview: string): { label: string; text: string; error: boolean }[] {
const { input, output } = parts(preview);
return [
{ label: "in", text: compact(input), error: false },
{ label: "out", text: compact(output), error: !!stepFailure(preview) },
].filter((line) => line.text);
}