From 7de044b71e6ea62da53ecec52854d8728ed28f4d Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 3 Oct 2026 17:22:13 -0700 Subject: [PATCH] feat(lens): read review spans as a conversation timeline Turns spans into the user's ask, tool calls with args and results, and the agent's reply, dropping system prompts. Also handles a preview cut that lands inside the Output header. --- .../components/lens/model/spanPreview.test.ts | 161 +++++++++++---- .../src/components/lens/model/spanPreview.ts | 190 +++++++++++++----- 2 files changed, 261 insertions(+), 90 deletions(-) diff --git a/ui/litellm-dashboard/src/components/lens/model/spanPreview.test.ts b/ui/litellm-dashboard/src/components/lens/model/spanPreview.test.ts index b9c70cd146b..64d6cf215cb 100644 --- a/ui/litellm-dashboard/src/components/lens/model/spanPreview.test.ts +++ b/ui/litellm-dashboard/src/components/lens/model/spanPreview.test.ts @@ -1,51 +1,126 @@ import { describe, expect, it } from "vitest"; -import { spanPreviewLines } from "./spanPreview"; +import { assistantReply, stepFailure, timeline, toolCall, userAsk } from "./spanPreview"; -describe("span previews", () => { - it("shows the last chat message of an agent span as role: content, not raw JSON", () => { - const preview = - 'Input: [{"role": "system", "content": "You are a research assistant."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find it."}]'; - expect(spanPreviewLines(preview)).toEqual([ - { label: "in", text: "user: Find the median latency of the EU region.", error: false }, - { label: "out", text: "assistant: I could not find it.", error: false }, - ]); +const OMIT = "\n[... preview omitted; read this span for evidence ...]\n"; + +const span = (kind: string, name: string, preview: string, cited = false) => ({ + span_id: `${kind}-${name}`, + kind, + name, + preview, + cited, +}); + +const agentPreview = + 'Input: [{"role": "system", "content": "You are a research assistant. Cite sources."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find a source for that."}]'; + +describe("user ask", () => { + it("takes the user's message and never the system prompt", () => { + expect(userAsk(agentPreview)).toBe("Find the median latency of the EU region."); }); - it("recovers the last message from truncated JSON with the omitted marker", () => { - const preview = - 'Input: [{"role": "system", "content": "You are a bill\n[... preview omitted; read this span for evidence ...]\nn"}, {"role": "user", "content": "Cancel C2473\'s subscript'; - expect(spanPreviewLines(preview)).toEqual([ - { label: "in", text: "user: Cancel C2473's subscript", error: false }, - ]); + it("marks a user message cut off by the preview limit", () => { + const cut = 'Input: [{"role": "system", "content": "You are a billing agent."}, {"role": "user", "content": "Cancel C2473\'s subscript'; + expect(userAsk(cut)).toBe("Cancel C2473's subscript…"); }); - it("keeps tool payloads compact and drops an unset status", () => { - const preview = 'Input: {"query":"latency"}\nOutput: {"results": []}\nStatus: STATUS_CODE_UNSET '; - expect(spanPreviewLines(preview)).toEqual([ - { label: "in", text: '{"query":"latency"}', error: false }, - { label: "out", text: '{"results":[]}', error: false }, - ]); - }); - - it("marks a failed span's output and shows its error", () => { - const preview = - 'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500 search index unavailable'; - const lines = spanPreviewLines(preview); - expect(lines.at(-1)).toEqual({ label: "error", text: "500 search index unavailable", error: true }); - expect(lines.find((l) => l.label === "out")?.error).toBe(true); - }); - - it("keeps multi-line payloads inside their section and collapses whitespace", () => { - const preview = 'Output: {"content": "def total(items):\\n return 1"}\nmore text\nStatus: STATUS_CODE_OK'; - expect(spanPreviewLines(preview)).toEqual([ - { label: "out", text: '{"content": "def total(items):\\n return 1"} more text', error: false }, - ]); - }); - - it("passes through plain text without sections", () => { - expect(spanPreviewLines("500 Internal Server Error")).toEqual([ - { label: "", text: "500 Internal Server Error", error: false }, - ]); - expect(spanPreviewLines("")).toEqual([]); + it("is null when only a system prompt is visible", () => { + expect(userAsk(`Input: [{"role": "system", "content": "You are a bill${OMIT}x"}]`)).toBeNull(); + }); +}); + +describe("assistant reply", () => { + it("reads the assistant output of an agent span", () => { + expect(assistantReply(agentPreview)).toBe("I could not find a source for that."); + }); + + it("recovers the end of a reply from a truncated llm span without its input", () => { + const llm = `Input: [{"role": "system", "content": "You are Acme's${OMIT}you want, I can also help you draft the message to your bank."}]\nStatus: STATUS_CODE_UNSET `; + expect(assistantReply(llm)).toBe("…you want, I can also help you draft the message to your bank."); + }); + + it("ignores a truncated tail that is a tool call or a JSON payload", () => { + const toolTail = `Input: [{"role": "system", "content": "You are a rese${OMIT}rch_docs", "arguments": "{\\"query\\":\\"latency\\"}"}}]}]\nStatus: STATUS_CODE_UNSET`; + expect(assistantReply(toolTail)).toBeNull(); + }); + + it("is null when the output is empty", () => { + expect(assistantReply('Input: [{"role": "user", "content": "hi"}]\nOutput: \nStatus: STATUS_CODE_ERROR x')).toBeNull(); + }); +}); + +describe("tool call", () => { + it("names the tool and keeps args and result compact", () => { + const tool = span( + "tool", + "execute_tool lookup_order", + 'Input: {"order_id":"A4160"}\nOutput: {"order_id": "A4160", "status": "processing"}\nStatus: STATUS_CODE_UNSET ', + ); + expect(toolCall(tool)).toEqual({ + kind: "tool", + name: "lookup_order", + args: '{"order_id":"A4160"}', + result: '{"order_id":"A4160","status":"processing"}', + error: false, + }); + }); + + it("flags error status and error payloads but not ordinary results", () => { + const failed = 'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500'; + expect(toolCall(span("tool", "execute_tool search_docs", failed)).error).toBe(true); + const denied = 'Input: {}\nOutput: {"status": "denied", "reason": "needs approval"}\nStatus: STATUS_CODE_UNSET'; + expect(toolCall(span("tool", "execute_tool issue_refund", denied)).error).toBe(true); + const tests = 'Input: {}\nOutput: {"passed": 41, "failed": 3}\nStatus: STATUS_CODE_UNSET'; + expect(toolCall(span("tool", "execute_tool run_tests", tests)).error).toBe(false); + }); + + it("reads a tool result that only survives in the preview tail", () => { + const cut = `Input: {"url":"https://example.com"}\nOutput: {"url": ${OMIT}, "text": "the figure is 37%"}\nStatus: STATUS_CODE_UNSET`; + expect(toolCall(span("tool", "execute_tool fetch_url", cut)).result).toBe('…, "text": "the figure is 37%"}'); + }); + + it("keeps args clean when the preview cut lands inside the Output header", () => { + const cut = `Input: {"customer_id":"C3847"}\nO${OMIT}t month", "amount_usd": 416.67}\nStatus: STATUS_CODE_UNSET `; + const call = toolCall(span("tool", "execute_tool get_invoice", cut)); + expect(call.args).toBe('{"customer_id":"C3847"}'); + expect(call.result).toBe('…t month", "amount_usd": 416.67}'); + }); +}); + +describe("step failure", () => { + it("reports only non-ok statuses", () => { + expect(stepFailure("Output: \nStatus: STATUS_CODE_ERROR Request timed out.")).toBe("Request timed out."); + expect(stepFailure("Output: x\nStatus: STATUS_CODE_UNSET ")).toBeNull(); + expect(stepFailure("plain")).toBeNull(); + }); +}); + +describe("timeline", () => { + it("reads like a conversation: ask, tool steps, then the final reply, without system text", () => { + const items = timeline([ + span("agent", "invoke_agent research-agent", agentPreview), + span("llm", "chat gpt", `Input: [{"role": "system", "content": "You are a rese${OMIT}I could not find a source for that."}]`), + span( + "tool", + "execute_tool search_docs", + 'Input: {"query":"EU latency"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500', + true, + ), + ]); + expect(items.map((i) => i.kind)).toEqual(["ask", "tool", "reply"]); + expect(items[0]).toMatchObject({ text: "Find the median latency of the EU region.", span: 0 }); + expect(items[1]).toMatchObject({ name: "search_docs", error: true, span: 2 }); + expect(items[2]).toMatchObject({ text: "I could not find a source for that.", span: 0 }); + expect(JSON.stringify(items)).not.toContain("research assistant"); + }); + + it("shows a failed model call as a failure step", () => { + const items = timeline([span("llm", "chat", "Input: x\nOutput: \nStatus: STATUS_CODE_ERROR Request timed out.")]); + expect(items).toEqual([{ kind: "failure", text: "Model call failed: Request timed out.", span: 0 }]); + }); + + it("falls back to short notes when nothing conversational can be recovered", () => { + const items = timeline([span("chain", "plan", 'Input: {"step": 1}')]); + expect(items).toEqual([{ kind: "note", label: "plan", text: '{"step":1}', span: 0 }]); }); }); diff --git a/ui/litellm-dashboard/src/components/lens/model/spanPreview.ts b/ui/litellm-dashboard/src/components/lens/model/spanPreview.ts index 216f20126ae..ae7a7e0d480 100644 --- a/ui/litellm-dashboard/src/components/lens/model/spanPreview.ts +++ b/ui/litellm-dashboard/src/components/lens/model/spanPreview.ts @@ -1,65 +1,161 @@ -import { parseJson, parseMessages, previewText } from "@/components/view_logs/TraceView/traceUtils"; +import { parseJson } from "@/components/view_logs/TraceView/traceUtils"; -export interface PreviewLine { - label: string; - text: string; - error: boolean; +import type { Review } from "./types"; + +type Span = Review["spans"][number]; + +export type TimelineItem = + | { kind: "ask"; text: string; span: number } + | { kind: "tool"; name: string; args: string; result: string; error: boolean; span: number } + | { kind: "reply"; text: string; span: number } + | { kind: "failure"; text: string; span: number } + | { kind: "note"; label: string; text: string; span: number }; + +interface Message { + role: string; + content: string; } -const SECTION = /^(Input|Output|Status): ?/; -const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/g; +const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/; +const CUT_OUTPUT_HEADER = /\nO(?:u(?:t(?:p(?:u(?:t:?)?)?)?)?)? ?$/; +const SECTION_START =/(?:^|\n)(Input|Output|Status): ?/g; const OK_STATUS = /^STATUS_CODE_(UNSET|OK)\b/; -const ROLE_CONTENT = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)/g; - -function sections(preview: string): Readonly> { - const lines = preview.split("\n"); - const starts = lines.flatMap((line, index) => (SECTION.test(line) ? [index] : [])); - if (!starts.length) return { Body: preview }; - return Object.fromEntries( - starts.map((start, n) => { - const body = lines.slice(start, starts[n + 1] ?? lines.length).join("\n"); - const name = SECTION.exec(body)?.[1] ?? "Body"; - return [name, body.replace(SECTION, "").trim()]; - }), - ); -} +const MESSAGE = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)(")?/g; +const TOOL_PROBLEM = /"error"|\bdenied\b|\bforbidden\b|\bunauthori[sz]ed\b|\bnot (?:allowed|permitted)\b/i; +const SNIPPET = 160; function decode(escaped: string): string { const parsed = parseJson(`"${escaped.replace(/\\u[0-9a-fA-F]{0,3}$|\\$/, "")}"`); return typeof parsed === "string" ? parsed : escaped; } -function lastMessage(value: string): string | null { - const messages = parseMessages(value); - if (messages?.length) { - const last = messages.at(-1); - return last ? `${last.role}: ${last.content}` : null; - } - const matches = [...value.matchAll(ROLE_CONTENT)]; - const match = matches.at(-1); - return match ? `${match[1]}: ${decode(match[2])}` : null; +function tidy(text: string): string { + return text.replace(/\s+/g, " ").trim(); } -function compactJson(value: string): string { - const parsed = parseJson(value); - if (parsed === null || typeof parsed !== "object") return value; - return JSON.stringify(parsed); +function sections(text: string): Readonly>> { + const starts = [...text.matchAll(SECTION_START)]; + return Object.fromEntries( + starts.map((match, n) => { + const from = (match.index ?? 0) + match[0].length; + const to = starts[n + 1]?.index ?? text.length; + return [match[1], text.slice(from, to)]; + }), + ); } -function readable(value: string): string { - const joined = value.replace(OMITTED, " … "); - return (lastMessage(joined) ?? previewText(compactJson(joined))).replace(/\s+/g, " ").trim(); +function parts(preview: string) { + const [rawHead, tail = ""] = preview.split(OMITTED); + const head = rawHead.replace(CUT_OUTPUT_HEADER, "\nOutput: "); + const whole = sections(preview.replace(OMITTED, "\n")); + const ending = sections(tail); + const before = sections(head); + const cutOutput = before.Output !== undefined ? `…${tail.split(/\nStatus: /)[0] ?? ""}` : ""; + return { + input: before.Input ?? "", + output: tail ? ending.Output ?? cutOutput : whole.Output ?? "", + status: (whole.Status ?? "").trim(), + tail: tail.split(/\nStatus: /)[0] ?? "", + truncated: !!tail, + }; } -export function spanPreviewLines(preview: string): PreviewLine[] { - const parts = sections(preview); - const status = parts.Status ?? ""; +function messages(text: string): Message[] { + return [...text.matchAll(MESSAGE)].map(([, role, content, closed]) => ({ + role, + content: tidy(decode(content)) + (closed ? "" : "…"), + })); +} + +function compact(text: string): string { + const trimmed = text.trim(); + const parsed = parseJson(trimmed); + const flat = parsed !== null && typeof parsed === "object" ? JSON.stringify(parsed) : trimmed; + return tidy(flat).slice(0, SNIPPET); +} + +export function stepFailure(preview: string): string | null { + const { status } = parts(preview); + if (!status || OK_STATUS.test(status)) return null; + return status.replace(/^STATUS_CODE_ERROR\s*/, "") || "failed"; +} + +export function userAsk(preview: string): string | null { + const asked = messages(parts(preview).input).filter((m) => m.role === "user" || m.role === "human"); + return asked.at(-1)?.content || null; +} + +function replyFromTail(tail: string): string | null { + const end = tail.trimEnd(); + if (!end.endsWith('"}]') || end.includes('\\"') || /"function"|\{"|":\s/.test(end)) return null; + const text = tidy(end.slice(0, -3)); + return text ? `…${text}` : null; +} + +export function assistantReply(preview: string): string | null { + const { output, tail, truncated } = parts(preview); + const said = messages(output).filter((m) => m.role === "assistant" && m.content && m.content !== "…"); + if (said.length) return said.at(-1)?.content ?? null; + return truncated && !output ? replyFromTail(tail) : null; +} + +export function toolCall(span: Pick): Omit, "span"> { + const { input, output, status } = parts(span.preview); + const result = compact(output); const failed = !!status && !OK_STATUS.test(status); - const content = [ - ...(parts.Body ? [{ label: "", text: readable(parts.Body), error: false }] : []), - ...(parts.Input ? [{ label: "in", text: readable(parts.Input), error: false }] : []), - ...(parts.Output ? [{ label: "out", text: readable(parts.Output), error: failed }] : []), - ...(failed ? [{ label: "error", text: status.replace(/^STATUS_CODE_ERROR\s*/, ""), error: true }] : []), - ]; - return content.filter((line) => line.text); + return { + kind: "tool", + name: span.name.replace(/^execute_tool\s+/, ""), + args: compact(input), + result: result || (failed ? stepFailure(span.preview) ?? "" : ""), + error: failed || TOOL_PROBLEM.test(output), + }; +} + +function overlaps(a: string, b: string): boolean { + const strip = (s: string) => s.replace(/…/g, "").trim(); + const [x, y] = [strip(a), strip(b)]; + return !!x && !!y && (x.includes(y) || y.includes(x)); +} + +function itemFor(span: Span, index: number): TimelineItem | null { + if (span.kind === "tool") return { ...toolCall(span), span: index }; + if (span.kind === "llm") { + const failure = stepFailure(span.preview); + if (failure) return { kind: "failure", text: `Model call failed: ${failure}`, span: index }; + const reply = assistantReply(span.preview); + return reply ? { kind: "reply", text: reply, span: index } : null; + } + return null; +} + +function fallback(spans: readonly Span[]): TimelineItem[] { + return spans.map((span, index) => { + const { input, output } = parts(span.preview); + const said = messages(input).filter((m) => m.role !== "system").at(-1)?.content; + return { kind: "note", label: span.name, text: said ?? compact(output || input), span: index }; + }); +} + +export function timeline(spans: readonly Span[]): TimelineItem[] { + const askAt = spans.findIndex((span) => userAsk(span.preview)); + const ask = askAt >= 0 ? userAsk(spans[askAt].preview) : null; + const agentAt = spans.findIndex((span) => span.kind === "agent" && assistantReply(span.preview)); + const final = agentAt >= 0 ? assistantReply(spans[agentAt].preview) : null; + const steps = spans.flatMap((span, index) => itemFor(span, index) ?? []); + const middle = final ? steps.filter((item) => item.kind !== "reply" || !overlaps(item.text, final)) : steps; + const items: TimelineItem[] = [ + ...(ask ? [{ kind: "ask" as const, text: ask, span: askAt }] : []), + ...middle, + ...(final ? [{ kind: "reply" as const, text: final, span: agentAt }] : []), + ]; + return items.length ? items : fallback(spans); +} + +export function spanPreviewLines(preview: string): { label: string; text: string; error: boolean }[] { + const { input, output } = parts(preview); + return [ + { label: "in", text: compact(input), error: false }, + { label: "out", text: compact(output), error: !!stepFailure(preview) }, + ].filter((line) => line.text); }