From 3c79db6122ad6c91d87b1fe6be8ee07df2794dd5 Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Tue, 6 Oct 2026 16:08:27 -0700 Subject: [PATCH] feat(lens-ui): replace the conversation view with a thread view (#44947) * feat(lens-ui): group a trace conversation into prompt, work and reply turns Co-Authored-By: Claude Opus 5.5 * test(lens-ui): cover thread turns, repeated prompts and subagent work Co-Authored-By: Claude Opus 5.5 * refactor(lens-ui): export conversation step renderers for reuse Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): add a Thread view with a folded Worked bar per turn Co-Authored-By: Claude Opus 5.5 * test(lens-ui): prove the Thread tab folds work and opens the exact step Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): add thread to the trace view routes Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): add a Thread tab to the run header Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): render the Thread view in the run body Co-Authored-By: Claude Opus 5.5 * fix(lens-ui): let the thread view retry steps that failed to load Co-Authored-By: Claude Opus 5.5 * test(lens-ui): cover retrying a failed step in the thread view Co-Authored-By: Claude Opus 5.5 * refactor(lens-ui): move step and message renderers into ConversationParts Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): show run errors and capture warnings in the thread view Co-Authored-By: Claude Opus 5.5 * test(lens-ui): cover failed tools, run errors, warnings and subagents in thread Co-Authored-By: Claude Opus 5.5 * fix(lens-ui): skip the missing replies warning when an answer was recorded Co-Authored-By: Claude Opus 5.5 * test(lens-ui): cover the missing replies warning with a recorded answer Co-Authored-By: Claude Opus 5.5 * refactor(lens-ui): drop conversation from the trace view routes Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): give the Steps and Thread tabs icons and drop Conversation Co-Authored-By: Claude Opus 5.5 * refactor(lens-ui): stop rendering the conversation view in the run body Co-Authored-By: Claude Opus 5.5 * test(lens-ui): switch the workspace drawer test to the Thread tab Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- .../lens/LensWorkspace.integration.test.tsx | 4 +- .../detail/conversation/ConversationParts.tsx | 169 +++++ .../TraceConversation.integration.test.tsx | 670 ------------------ .../detail/conversation/TraceConversation.tsx | 398 ----------- .../TraceThread.integration.test.tsx | 253 +++++++ .../detail/conversation/TraceThread.tsx | 258 +++++++ .../detail/conversation/conversation.test.ts | 16 + .../detail/conversation/conversation.ts | 34 +- .../traces/detail/conversation/thread.test.ts | 194 +++++ .../lens/traces/detail/conversation/thread.ts | 173 +++++ .../lens/traces/detail/run/RunBody.tsx | 21 +- .../lens/traces/detail/run/RunHeader.tsx | 10 +- .../src/components/lens/traces/routing.ts | 2 +- 13 files changed, 1104 insertions(+), 1098 deletions(-) create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/ConversationParts.tsx delete mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.integration.test.tsx delete mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.integration.test.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.test.ts create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.ts diff --git a/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx index f42f2bd7e8c..c251b36d190 100644 --- a/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx @@ -85,8 +85,8 @@ describe("Lens interactive demo", () => { expect(lastUrl(onUrlUpdate).get("trace")).toBe(openRun.get("trace")); await user.click(within(drawer).getByRole("tab", { name: "Attributes" })); await expectUrl(onUrlUpdate, (url) => expect(url.get("span_tab")).toBe("attributes")); - await user.click(within(drawer).getByRole("tab", { name: "Conversation" })); - await expectUrl(onUrlUpdate, (url) => expect(url.get("view")).toBe("conversation")); + await user.click(within(drawer).getByRole("tab", { name: "Thread" })); + await expectUrl(onUrlUpdate, (url) => expect(url.get("view")).toBe("thread")); expect(network).not.toHaveBeenCalled(); await user.click(screen.getByRole("switch", { name: "Demo data" })); expect(await screen.findByRole("region", { name: "Get started with Lens" })).toBeVisible(); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/ConversationParts.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/ConversationParts.tsx new file mode 100644 index 00000000000..2621568b397 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/ConversationParts.tsx @@ -0,0 +1,169 @@ +"use client"; + +import { ChevronRight, Wrench } from "lucide-react"; +import { useState } from "react"; + +import CopyButton from "@/components/shared/CopyButton"; +import { Button } from "@/components/ui/button"; +import { cn } from "@/lib/cva.config"; + +import type { TraceMessage } from "../../types"; +import { fmtMs } from "../../utils"; +import { Markdown } from "../content/Markdown"; +import { ToolCallBlock, ToolResultCard } from "../content/Messages"; +import { toolSummary } from "../content/payload"; +import { ErrorBlock } from "../content/SpanError"; +import { ToolArguments, ToolOutput } from "../content/ToolContent"; +import type { ConversationGroup, ConversationItem } from "./conversation"; + +export function ConversationSteps({ + groups, + onOpenStep, +}: { + groups: readonly ConversationGroup[]; + onOpenStep: (id: string) => void; +}) { + return groups.map((group) => + group.kind === "item" ? ( + + ) : ( +
+ Subagent: {group.name} +
+ +
+
+ ), + ); +} + +export function ConversationStep({ item, onOpenStep }: { item: ConversationItem; onOpenStep: (id: string) => void }) { + return ( +
+ {item.toolResult === undefined && ( +
+
+ {item.model && ( + + {item.model} + + )} + {fmtMs(item.time ?? item.span.start_offset_ms)} +
+ +
+ )} + {item.showError && } + {item.messages.map((message, index) => ( + + ))} + {item.toolResult !== undefined && } +
+ ); +} + +function ConversationTool({ item, onOpenStep }: { item: ConversationItem; onOpenStep: (id: string) => void }) { + const [open, setOpen] = useState(item.span.status === "error"); + const failed = item.span.status === "error"; + const summary = toolSummary(item.toolCall?.args); + return ( +
+
+ + +
+ {open && ( +
+ {item.toolCall && } + {item.span.error && item.span.error !== item.toolResult && ( +

{item.span.error}

+ )} +
+ Result + +
+ {item.toolResult ? ( + + ) : ( +

No result content recorded.

+ )} +
+ )} +
+ ); +} + +export function ConversationMessage({ message }: { message: TraceMessage }) { + if (message.role === "system" && (message.content.startsWith("Agent ") || message.content === "Context compacted")) + return

{message.content}

; + if (message.role === "system") + return ( +
+ Session context +
+ +
+
+ ); + if (message.role === "tool") return ; + return ( +
+ {message.content && ( +
+ +
+ )} + {message.tool_calls?.map((call, index) => ( +
+ {call.name} +
+ +
+
+ ))} +
+ ); +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.integration.test.tsx deleted file mode 100644 index 884cd0d5304..00000000000 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.integration.test.tsx +++ /dev/null @@ -1,670 +0,0 @@ -import { act, screen, waitFor, within } from "@testing-library/react"; -import userEvent from "@testing-library/user-event"; -import { beforeEach, describe, expect, it, vi } from "vitest"; -import type { ComponentProps } from "react"; -import { renderWithProviders, testQueryClient } from "../../../../../../tests/test-utils"; -import { RunView } from "../run/RunView"; -import { useOpenTraceRouting } from "../../routing"; -import { TraceConversation } from "./TraceConversation"; -import type { SpanDetail, Trace } from "../../types"; -import research from "../../__fixtures__/research_trace.json"; - -vi.mock("../../../../networking", () => ({ - agentTraceCall: vi.fn(), - agentTraceSpanCall: vi.fn(), - getProxyBaseUrl: () => "http://proxy.test", -})); -import { agentTraceCall, agentTraceSpanCall } from "../../../../networking"; - -function RoutedRunView(props: Omit, "selection">) { - const { selection } = useOpenTraceRouting(); - return ; -} - -const root = { ...research.spans[0], span_id: "root", parent_span_id: null }; -const tool = { ...root, span_id: "tool", name: "read_file", parent_span_id: "root", type: "tool", start_offset_ms: 1 }; -const trace = { ...research, spans: [root, tool] } as Trace; -const rootDetail: SpanDetail = { - span_id: "root", - input: '[{"role":"user","content":"Read the release notes"}]', - output: '[{"role":"assistant","content":"The release is ready"}]', - attributes: {}, -}; -const toolDetail: SpanDetail = { - span_id: "tool", - input: '{"path":"CHANGELOG.md"}', - output: "All checks passed", - attributes: {}, -}; - -describe("TraceConversation", () => { - beforeEach(() => { - testQueryClient.clear(); - vi.mocked(agentTraceCall).mockReset().mockResolvedValue(trace); - vi.mocked(agentTraceSpanCall) - .mockReset() - .mockImplementation(async (_token, _trace, id) => (id === "root" ? rootDetail : { ...toolDetail, span_id: id })); - }); - - it.each([false, true])("refreshes unchanged spans without hiding loaded content on failure (%s)", async (failed) => { - const user = userEvent.setup(); - vi.mocked(agentTraceCall).mockResolvedValue({ - ...trace, - summary: { ...trace.summary, trace_ref: "resolved-reference" }, - }); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("The release is ready")).toBeVisible(); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (failed) throw new Error("content refresh failed"); - return id === "root" ? { ...rootDetail, output: "Updated final answer" } : toolDetail; - }); - await user.click(screen.getByRole("button", { name: "Refresh run" })); - if (failed) { - expect(await screen.findAllByRole("button", { name: "Retry step" })).toHaveLength(2); - expect(screen.getByText("The release is ready")).toBeVisible(); - } else { - expect(await screen.findByText("Updated final answer")).toBeVisible(); - expect(screen.queryByText("The release is ready")).not.toBeInTheDocument(); - } - }); - - it("switches to a readable transcript and opens the exact tool step from it", async () => { - const user = userEvent.setup(); - renderWithProviders( - , - ); - const conversationTab = await screen.findByRole("tab", { name: "Conversation", selected: false }); - await user.click(conversationTab); - expect(conversationTab).toHaveAttribute("aria-selected", "true"); - const conversation = await screen.findByRole("region", { name: "Trace conversation" }); - expect(await within(conversation).findByText("Read the release notes")).toBeVisible(); - expect(await within(conversation).findByRole("button", { name: "Expand read_file tool call" })).toHaveTextContent( - "CHANGELOG.md", - ); - await user.click(within(conversation).getByRole("button", { name: "Expand read_file tool call" })); - expect(await within(conversation).findByText("CHANGELOG.md")).toBeVisible(); - expect(await within(conversation).findByText("All checks passed")).toBeVisible(); - expect(await within(conversation).findByText("The release is ready")).toBeVisible(); - const toolStep = within(conversation).getByRole("region", { name: "Conversation step read_file" }); - await user.click(within(toolStep).getByRole("button", { name: "Inspect step read_file" })); - expect(screen.getByRole("tab", { name: "Steps", selected: true })).toBeVisible(); - expect(screen.getByRole("treeitem", { selected: true })).toHaveAttribute("data-row-id", "tool"); - expect(screen.getByRole("heading", { name: "read_file" })).toBeVisible(); - }); - - it("waits for an in-flight refresh before requesting another conversation page", async () => { - const user = userEvent.setup(); - const first = { ...trace, spans: [root], next_cursor: "old-page" }; - const refreshed = { ...first, next_cursor: "fresh-page" }; - const second = { ...trace, spans: [tool], next_cursor: null }; - const pending = Promise.withResolvers(); - vi.mocked(agentTraceCall) - .mockResolvedValueOnce(first) - .mockReturnValueOnce(pending.promise) - .mockResolvedValue(second); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("Read the release notes")).toBeVisible(); - await user.click(screen.getByRole("button", { name: "Refresh run" })); - const more = screen.getByRole("button", { name: "Load next 20 entries" }); - expect(more).toBeDisabled(); - await user.click(more); - expect(agentTraceCall).toHaveBeenCalledTimes(2); - await act(async () => pending.resolve(refreshed)); - await waitFor(() => expect(more).toBeEnabled()); - await user.click(more); - expect(await screen.findByRole("button", { name: "Expand read_file tool call" })).toBeVisible(); - expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([null, null, "fresh-page"]); - }); - - it("can load the next conversation page after a refresh fails without invalidating loaded content", async () => { - const user = userEvent.setup(); - const first = { ...trace, spans: [root], next_cursor: "next-page" }; - const second = { ...trace, spans: [tool], next_cursor: null }; - vi.mocked(agentTraceCall) - .mockResolvedValueOnce(first) - .mockRejectedValueOnce(new Error("refresh unavailable")) - .mockResolvedValue(second); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("Read the release notes")).toBeVisible(); - const contentReads = vi.mocked(agentTraceSpanCall).mock.calls.length; - vi.mocked(agentTraceSpanCall).mockRejectedValue(new Error("content unavailable")); - await user.click(screen.getByRole("button", { name: "Refresh run" })); - expect(await screen.findByRole("alert")).toHaveTextContent("Previously received steps are still shown"); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(contentReads); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => - id === "root" ? rootDetail : toolDetail, - ); - const more = screen.getByRole("button", { name: "Load next 20 entries" }); - expect(more).toBeEnabled(); - await user.click(more); - expect(await screen.findByRole("button", { name: "Expand read_file tool call" })).toBeVisible(); - expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([null, null, "next-page"]); - expect(screen.queryByText(/Could not load more conversation entries/)).not.toBeInTheDocument(); - }); - - it("keeps refresh busy until content finishes when live updates pause", async () => { - const user = userEvent.setup(); - const pending = Promise.withResolvers(); - vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, next_cursor: "next-page" }); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("The release is ready")).toBeVisible(); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => - id === "root" ? pending.promise : toolDetail, - ); - const refresh = screen.getByRole("button", { name: "Refresh run" }); - await user.click(refresh); - await waitFor(() => expect(testQueryClient.isFetching({ queryKey: ["agentTrace"] })).toBe(0)); - expect(refresh).toBeDisabled(); - const more = screen.getByRole("button", { name: "Load next 20 entries" }); - expect(more).toBeDisabled(); - await user.click(refresh); - await user.click(screen.getByRole("button", { name: "Live updates" })); - await act(async () => pending.resolve({ ...rootDetail, output: "Updated final answer" })); - expect(await screen.findByText("Updated final answer")).toBeVisible(); - await waitFor(() => expect(refresh).toBeEnabled()); - expect(more).toBeEnabled(); - expect(agentTraceCall).toHaveBeenCalledTimes(2); - }); - - it("renders a failed shell exchange in both views and preserves its raw result", async () => { - const user = userEvent.setup(); - const command = "npm test -- checkout\nprintf 'finished\\n'"; - const output = JSON.stringify({ output: "PASS cart.test.ts\nFAIL checkout.test.ts", exit_code: 1, error: null }); - const failedTool = { ...tool, name: "terminal", status: "error", error: null }; - vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root, failedTool] } as Trace); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => - id === "root" - ? rootDetail - : { - ...toolDetail, - input: JSON.stringify({ command, workdir: "/workspace" }), - output, - input_ui: { - kind: "fields", - fields: [ - { key: "command", value: command }, - { key: "workdir", value: "/workspace" }, - ], - }, - output_ui: { - kind: "fields", - fields: [ - { key: "output", value: "PASS cart.test.ts\nFAIL checkout.test.ts" }, - { key: "exit_code", value: "1" }, - ], - }, - }, - ); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - const conversation = await screen.findByRole("region", { name: "Trace conversation" }); - expect(await within(conversation).findByText(/npm test -- checkout/, { selector: "pre" })).toHaveTextContent( - "printf 'finished\\n'", - ); - expect(within(conversation).getByText(/PASS cart.test.ts/, { selector: "pre" })).toHaveTextContent( - "FAIL checkout.test.ts", - ); - expect(within(conversation).getByText("exit_code")).toBeVisible(); - await user.click(within(conversation).getByRole("button", { name: "Inspect step terminal" })); - const details = screen.getByRole("complementary", { name: "Span details" }); - expect(await within(details).findByText(/npm test -- checkout/, { selector: "pre" })).toBeVisible(); - const result = within(details).getByRole("region", { name: "Output", exact: true }); - expect(within(result).getByText("exit_code")).toBeVisible(); - await user.click(within(result).getByRole("radio", { name: "Raw" })); - expect(within(result).getByText(/"exit_code": 1/, { selector: "pre" })).toHaveTextContent('"error": null'); - }); - - it("loads twenty span details initially and pages conversation entries on demand", async () => { - const user = userEvent.setup(); - const spans = [ - root, - ...Array.from({ length: 30 }, (_, index) => ({ ...tool, span_id: `tool-${index}`, start_offset_ms: index + 1 })), - ]; - const long = { ...trace, spans } as Trace; - renderWithProviders(); - const more = await screen.findByRole("button", { name: "Load next 20 entries" }); - await waitFor(() => expect(more).toBeEnabled()); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(20); - expect(screen.queryByText("The release is ready")).not.toBeInTheDocument(); - await user.click(more); - expect(await screen.findByText("The release is ready")).toBeVisible(); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(31); - expect(screen.queryByRole("button", { name: /Load next/ })).not.toBeInTheDocument(); - }); - - it("shows a loaded agent reply while later trace pages remain available", async () => { - const user = userEvent.setup(); - vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root], next_cursor: "next-page" }); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("The release is ready")).toBeVisible(); - expect(screen.getByRole("button", { name: "Load more steps" })).toBeVisible(); - expect(screen.queryByText("End of conversation")).not.toBeInTheDocument(); - }); - - it("adds twenty visible entries across hidden subagents, supplemental spans, and server pages", async () => { - const user = userEvent.setup(); - const child = { ...root, span_id: "child", parent_span_id: "root", name: "Reviewer", start_offset_ms: 1 }; - const childSpans = Array.from({ length: 50 }, (_, index) => [ - { - ...tool, - span_id: `child-${index}`, - parent_span_id: "child", - name: `child tool ${index}`, - start_offset_ms: 2 + index * 2, - }, - { - ...tool, - span_id: `supplement-${index}`, - parent_span_id: "child", - name: "claude_code.tool_result", - framework: "claude-code", - type: "event", - start_offset_ms: 3 + index * 2, - }, - ]).flat(); - const parentSpans = Array.from({ length: 45 }, (_, index) => [ - { ...tool, span_id: `parent-${index}`, name: `parent tool ${index}`, start_offset_ms: 102 + index * 2 }, - { ...tool, span_id: `empty-${index}`, type: "llm", start_offset_ms: 103 + index * 2 }, - ]).flat(); - const spans = [root, child, ...childSpans, ...parentSpans] as Trace["spans"]; - const first: Trace = { ...trace, spans: spans.slice(0, 120), next_cursor: "next-page" }; - const last: Trace = { ...trace, spans: spans.slice(120), next_cursor: null }; - vi.mocked(agentTraceCall).mockImplementation(async (_token, _trace, _ref, cursor) => (cursor ? last : first)); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "root" || id === "child") - return { ...rootDetail, span_id: id, input: id === "root" ? "Main task" : "Child task", output: "" }; - if (id.startsWith("empty-") || id.startsWith("supplement-")) - return { span_id: id, input: "", output: "", attributes: {} }; - return { ...toolDetail, span_id: id }; - }); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("2 entries shown")).toBeVisible(); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(20); - await user.click(screen.getByRole("button", { name: "Load next 20 entries", exact: true })); - expect(await screen.findByText("22 entries shown")).toBeVisible(); - expect(screen.getAllByRole("region", { name: /Conversation step parent tool/ })).toHaveLength(20); - expect( - screen.queryByRole("button", { name: "Expand parent tool 20 tool call", exact: true }), - ).not.toBeInTheDocument(); - expect(screen.getByRole("region", { name: "Conversation step child tool 0", exact: true })).not.toBeVisible(); - expect(agentTraceCall).toHaveBeenCalledWith("test", trace.summary.trace_id, undefined, "next-page"); - - await user.click(screen.getByText("Subagent: Reviewer", { exact: true })); - expect(screen.getAllByRole("region", { name: /Conversation step child tool/ })).toHaveLength(19); - await user.click(screen.getByRole("button", { name: "Load next 20 entries in Reviewer", exact: true })); - expect(screen.getAllByRole("region", { name: /Conversation step child tool/ })).toHaveLength(39); - expect(screen.getAllByRole("region", { name: /Conversation step parent tool/ })).toHaveLength(20); - expect(screen.queryByText("End of conversation")).not.toBeInTheDocument(); - }); - - it.each([ - { duration: 10, nextCursor: null }, - { duration: 1000, nextCursor: null }, - { duration: 10, nextCursor: "unrelated-page" }, - ])("hides an exhausted subagent control with unrelated work remaining (%j)", async ({ duration, nextCursor }) => { - const user = userEvent.setup(); - const child = { - ...root, - span_id: "child", - parent_span_id: "root", - name: "Reviewer", - start_offset_ms: 1, - duration_ms: duration, - }; - const spans = [ - root, - child, - { ...tool, span_id: "child-tool", parent_span_id: "child", start_offset_ms: 2, duration_ms: 1 }, - ...Array.from({ length: 40 }, (_, index) => ({ - ...tool, - span_id: `parent-${index}`, - start_offset_ms: 100 + index, - duration_ms: 1, - })), - ] as Trace["spans"]; - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "child") return { ...rootDetail, span_id: id, input: "Review task", output: "Review finished" }; - return id === "root" ? rootDetail : { ...toolDetail, span_id: id }; - }); - const loadMore = vi.fn(); - renderWithProviders( - , - ); - const more = await screen.findByRole("button", { name: "Load next 20 entries", exact: true }); - await waitFor(() => expect(more).toBeEnabled()); - await user.click(screen.getByText("Subagent: Reviewer", { exact: true })); - expect(screen.getByText("Review finished")).toBeVisible(); - expect(screen.queryByRole("button", { name: /entries in Reviewer/ })).not.toBeInTheDocument(); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(20); - expect(loadMore).not.toHaveBeenCalled(); - }); - - it.each([false, true])("stops loading at the selected subagent's end (server paging: %s)", async (paged) => { - const user = userEvent.setup(); - const child = { - ...root, - span_id: "child", - parent_span_id: "root", - name: "Reviewer", - start_offset_ms: 1, - duration_ms: 60, - }; - const spans = [ - root, - child, - ...Array.from({ length: 25 }, (_, index) => ({ - ...tool, - span_id: `child-${index}`, - parent_span_id: "child", - name: `child tool ${index}`, - start_offset_ms: 2 + index, - duration_ms: 1, - })), - ...Array.from({ length: 40 }, (_, index) => ({ - ...tool, - span_id: `parent-${index}`, - name: `parent tool ${index}`, - start_offset_ms: 100 + index, - duration_ms: 1, - })), - ] as Trace["spans"]; - const first: Trace = { - ...trace, - spans: paged ? spans.slice(0, 20) : spans, - next_cursor: paged ? "child-page" : null, - }; - const next: Trace = { ...trace, spans: spans.slice(20, 40), next_cursor: "unrelated-page" }; - vi.mocked(agentTraceCall).mockImplementation(async (_token, _trace, _ref, cursor) => (cursor ? next : first)); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "child") return { ...rootDetail, span_id: id, input: "Review task", output: "Review finished" }; - return id === "root" ? rootDetail : { ...toolDetail, span_id: id }; - }); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - await user.click(await screen.findByText("Subagent: Reviewer", { exact: true })); - const more = await screen.findByRole("button", { name: "Load next 20 entries in Reviewer" }); - await waitFor(() => expect(more).toBeEnabled()); - await user.click(more); - expect(await screen.findByRole("button", { name: "Expand child tool 24 tool call" })).toBeVisible(); - await waitFor(() => expect(screen.queryByRole("button", { name: /entries in Reviewer/ })).not.toBeInTheDocument()); - expect(screen.getByText("Review finished")).toBeVisible(); - expect(screen.getAllByRole("region", { name: /Conversation step child tool/ })).toHaveLength(25); - expect(screen.getByRole("button", { name: "Load next 20 entries", exact: true })).toBeEnabled(); - expect(screen.queryByText("Loading conversation…")).not.toBeInTheDocument(); - expect(agentTraceSpanCall).toHaveBeenCalledTimes(40); - expect(agentTraceCall).toHaveBeenCalledTimes(paged ? 2 : 1); - expect(agentTraceCall).not.toHaveBeenCalledWith("test", trace.summary.trace_id, undefined, "unrelated-page"); - }); - - it("stops at a failed server page and resumes the requested entries after retry", async () => { - const user = userEvent.setup(); - const first: Trace = { ...trace, next_cursor: "next-page" }; - const last: Trace = { - ...trace, - spans: [{ ...tool, span_id: "later", name: "later tool", start_offset_ms: 2 }], - next_cursor: null, - }; - vi.mocked(agentTraceCall) - .mockResolvedValueOnce(first) - .mockRejectedValueOnce(new Error("temporarily unavailable")) - .mockResolvedValue(last); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - const more = await screen.findByRole("button", { name: "Load next 20 entries", exact: true }); - await waitFor(() => expect(more).toBeEnabled()); - await user.click(more); - expect(await screen.findByRole("alert")).toHaveTextContent("Could not load more conversation entries"); - expect(more).toBeDisabled(); - expect(agentTraceCall).toHaveBeenCalledTimes(2); - expect(screen.getByText("Read the release notes")).toBeVisible(); - await user.click(screen.getByRole("button", { name: "Retry", exact: true })); - expect(await screen.findByRole("button", { name: "Expand later tool tool call" })).toBeVisible(); - expect(await screen.findByText("End of conversation")).toBeVisible(); - expect(screen.queryByRole("alert")).not.toBeInTheDocument(); - }); - - it.each([false, true])( - "checks for missing replies after the final trace page and details load (reply: %s)", - async (hasReply) => { - const user = userEvent.setup(); - const laterDetail = Promise.withResolvers(); - const summary = { ...trace.summary, span_count: 2 }; - const first: Trace = { ...trace, summary, spans: [root], next_cursor: "last-page" }; - const last: Trace = { - ...trace, - summary, - spans: [{ ...tool, span_id: "later", name: "later response", type: "llm" }], - next_cursor: null, - }; - vi.mocked(agentTraceCall).mockImplementation(async (_token, _trace, _ref, cursor) => (cursor ? last : first)); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => - id === "root" ? { ...rootDetail, output: "", attributes: { "span.type": "llm_request" } } : laterDetail.promise, - ); - renderWithProviders( - , - ); - await user.click(await screen.findByRole("tab", { name: "Conversation" })); - expect(await screen.findByText("Read the release notes")).toBeVisible(); - expect(screen.queryByText(/no recorded assistant replies/)).not.toBeInTheDocument(); - expect(screen.queryByText("End of conversation")).not.toBeInTheDocument(); - await user.click(screen.getByRole("button", { name: "Load more steps" })); - expect(await screen.findByText("Loading conversation…")).toBeVisible(); - expect(screen.queryByText(/no recorded assistant replies/)).not.toBeInTheDocument(); - expect(screen.queryByText("End of conversation")).not.toBeInTheDocument(); - const resolved: SpanDetail = { - span_id: "later", - input: "", - output: hasReply ? rootDetail.output : "", - attributes: hasReply ? { "event.name": "assistant_response" } : {}, - }; - await act(async () => laterDetail.resolve(resolved)); - expect(await screen.findByText("End of conversation")).toBeVisible(); - if (hasReply) { - expect(screen.getByText("The release is ready")).toBeVisible(); - expect(screen.queryByText(/no recorded assistant replies/)).not.toBeInTheDocument(); - } else { - expect(screen.getByText(/no recorded assistant replies/)).toBeVisible(); - } - }, - ); - - it("shows distinct agent invocation labels together with each step's time", async () => { - const first = { ...root, name: "reviewer", start_offset_ms: 1000 }; - const second = { ...first, span_id: "second", start_offset_ms: 2000 }; - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => ({ - ...rootDetail, - span_id: id, - input: id === "root" ? "Review code" : "Review tests", - output: "", - })); - renderWithProviders( - , - ); - expect(await screen.findByText("reviewer (1)")).toBeVisible(); - expect(screen.getByText("reviewer (2)")).toBeVisible(); - const steps = screen.getAllByRole("region", { name: "Conversation step reviewer" }); - expect(within(steps[0]).getByText("1.00s")).toBeVisible(); - expect(within(steps[1]).getByText("2.00s")).toBeVisible(); - }); - - it.each([rootDetail.output, ""])( - "keeps the root failure visible while loading, then places it in order (output: %s)", - async (output) => { - const user = userEvent.setup(); - const rootFetch = Promise.withResolvers(); - const failedRoot = { ...root, status: "error", error: "Agent exceeded its execution limit" }; - const spans = [ - failedRoot, - ...Array.from({ length: 24 }, (_, index) => ({ - ...tool, - span_id: `tool-${index}`, - start_offset_ms: index + 1, - })), - ]; - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => - id === "root" ? rootFetch.promise : { ...toolDetail, span_id: id }, - ); - renderWithProviders( - , - ); - expect(screen.getAllByText("Agent exceeded its execution limit")).toHaveLength(1); - await act(async () => rootFetch.resolve({ ...rootDetail, output })); - const more = screen.getByRole("button", { name: "Load next 20 entries" }); - await waitFor(() => expect(more).toBeEnabled()); - expect(screen.getAllByText("Agent exceeded its execution limit")).toHaveLength(1); - expect(screen.queryByText("The release is ready")).not.toBeInTheDocument(); - await user.click(more); - expect(await screen.findByText("End of conversation")).toBeVisible(); - expect(screen.getAllByText("Agent exceeded its execution limit")).toHaveLength(1); - const rootSteps = screen.getAllByRole("region", { name: `Conversation step ${root.name}` }); - expect(within(rootSteps.at(-1)!).getByText("Agent exceeded its execution limit")).toBeVisible(); - }, - ); - - it("shows a missing step explicitly and lets the user retry it", async () => { - const user = userEvent.setup(); - vi.mocked(agentTraceSpanCall).mockRejectedValueOnce(new Error("temporarily unavailable")); - renderWithProviders(); - expect(await screen.findByRole("alert")).toHaveTextContent("Retry this step to continue the conversation"); - await user.click(screen.getByRole("button", { name: "Retry step" })); - expect(await screen.findByText("Read the release notes")).toBeVisible(); - expect(screen.queryByRole("alert")).not.toBeInTheDocument(); - }); - - it("pauses a requested page at a failed prefetched detail until it is retried", async () => { - const user = userEvent.setup(); - const spans = [ - root, - ...Array.from({ length: 45 }, (_, index) => ({ - ...tool, - span_id: `tool-${index}`, - name: `check ${index}`, - start_offset_ms: index + 1, - })), - ] as Trace["spans"]; - const retry = vi.fn().mockRejectedValueOnce(new Error("unavailable")).mockResolvedValue(toolDetail); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "tool-21") return retry(); - return id === "root" ? rootDetail : { ...toolDetail, span_id: id }; - }); - renderWithProviders(); - const more = screen.getByRole("button", { name: "Load next 20 entries", exact: true }); - await waitFor(() => expect(more).toBeEnabled()); - await user.click(more); - expect(await screen.findByRole("alert")).toHaveTextContent("Could not load check 21"); - expect(retry).toHaveBeenCalledTimes(1); - expect(screen.queryByRole("button", { name: "Expand check 22 tool call" })).not.toBeInTheDocument(); - await user.click(screen.getByRole("button", { name: "Retry step" })); - expect(await screen.findByText("40 entries shown")).toBeVisible(); - expect(screen.getAllByRole("region", { name: /Conversation step check/ })).toHaveLength(39); - expect(screen.queryByRole("alert")).not.toBeInTheDocument(); - }); - - it("holds later turns and the final answer behind a failed step until retry succeeds", async () => { - const user = userEvent.setup(); - const first = { ...tool, span_id: "first", name: "First response", type: "llm", start_offset_ms: 1 }; - const failedTool = { ...tool, start_offset_ms: 2 }; - const last = { ...first, span_id: "last", name: "Final response", start_offset_ms: 3 }; - const traced = { ...trace, spans: [root, first, failedTool, last] } as Trace; - const question = { role: "user", content: "Read the release notes" }; - const checking = { - role: "assistant", - content: "Checking the release", - tool_calls: [{ name: "read_file", args: { path: "CHANGELOG.md" } }], - }; - const firstDetail = { ...rootDetail, span_id: "first", output: JSON.stringify([checking]) }; - const lastDetail = { - ...rootDetail, - span_id: "last", - input: JSON.stringify([question, checking, { role: "tool", content: "All checks passed" }]), - }; - const toolFetch = vi.fn().mockRejectedValueOnce(new Error("unavailable")).mockResolvedValue(toolDetail); - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "tool") return toolFetch(); - if (id === "first") return firstDetail; - if (id === "last") return lastDetail; - return rootDetail; - }); - renderWithProviders(); - - expect(await screen.findByRole("alert")).toHaveTextContent("read_file"); - expect(await screen.findByText("Checking the release")).toBeVisible(); - expect(screen.queryByText("The release is ready")).not.toBeInTheDocument(); - expect(screen.queryByText("End of conversation")).not.toBeInTheDocument(); - expect(screen.getByText("2 entries shown")).toBeVisible(); - - await user.click(screen.getByRole("button", { name: "Retry step" })); - expect(await screen.findByText("The release is ready")).toBeVisible(); - expect(screen.getAllByText("Checking the release")).toHaveLength(1); - expect(screen.getAllByText("Read the release notes")).toHaveLength(1); - expect(screen.getAllByRole("button", { name: "Expand read_file tool call" })).toHaveLength(1); - expect(screen.getByText("End of conversation")).toBeVisible(); - }); - - it.each(["Partial investigation", ""])( - "shows a failed child agent's error once after its work, with output %j", - async (output) => { - const user = userEvent.setup(); - const agent = { - ...root, - span_id: "child", - parent_span_id: "root", - name: "Investigate release", - type: "agent", - start_offset_ms: 1, - duration_ms: 10, - status: "error", - error: "Investigation timed out", - }; - const childTool = { ...tool, parent_span_id: "child", start_offset_ms: 2, duration_ms: 2 }; - const traced = { ...trace, spans: [root, agent, childTool] } as Trace; - vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { - if (id === "child") return { ...rootDetail, span_id: id, input: "Investigate failed checks", output }; - return id === "root" ? rootDetail : toolDetail; - }); - renderWithProviders(); - - expect(await screen.findByText("End of conversation")).toBeVisible(); - await user.click(screen.getByText("Subagent: Investigate release", { exact: true })); - expect(screen.getAllByText("Investigation timed out")).toHaveLength(1); - const entries = screen.getAllByRole("region", { name: "Conversation step Investigate release" }); - expect(within(entries[0]).getByText("Investigate failed checks")).toBeVisible(); - expect(within(entries[0]).queryByText("Investigation timed out")).not.toBeInTheDocument(); - expect(within(entries[1]).getByText("Investigation timed out")).toBeVisible(); - if (output) expect(within(entries[1]).getByText(output)).toBeVisible(); - }, - ); -}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx deleted file mode 100644 index 93fb6456d28..00000000000 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx +++ /dev/null @@ -1,398 +0,0 @@ -"use client"; - -import { useEffect, useState } from "react"; -import { ChevronRight, Wrench } from "lucide-react"; -import { cn } from "@/lib/cva.config"; -import CopyButton from "@/components/shared/CopyButton"; -import { ToolArguments, ToolOutput } from "../content/ToolContent"; -import { toolSummary } from "../content/payload"; -import { Button } from "@/components/ui/button"; -import { - buildConversation, - conversationWarnings, - conversationBranchGroups, - pendingConversationBranches, - groupConversation, - CONVERSATION_PAGE_SIZE, - type ConversationItem, - type ConversationGroup, -} from "./conversation"; -import { ErrorBlock } from "../content/SpanError"; -import { Markdown } from "../content/Markdown"; -import { ToolCallBlock, ToolResultCard } from "../content/Messages"; -import type { Trace, TraceMessage } from "../../types"; -import { fmtMs } from "../../utils"; -import { useConversationDetails } from "./useConversationDetails"; - -export interface ConversationTracePaging { - loading: boolean; - failed: boolean; - loadMore: () => void; -} - -function hasRemainingSource(branchId: string | null, sourceRemaining: boolean, pendingBranches: ReadonlySet) { - return sourceRemaining && (branchId === null || pendingBranches.has(branchId)); -} - -export function TraceConversation({ - trace, - accessToken, - onOpenStep, - paging, -}: { - trace: Trace; - accessToken: string; - onOpenStep: (id: string) => void; - paging?: ConversationTracePaging; -}) { - const [pages, setPages] = useState>(new Map()); - const [requestedBranch, setRequestedBranch] = useState(); - const { details, entries, complete, loading, failed, hasMore, loadMore } = useConversationDetails(trace, accessToken); - const traceComplete = complete && !trace.next_cursor; - const pendingBranches = pendingConversationBranches(trace.spans, details, Boolean(trace.next_cursor)); - const items = buildConversation(trace.spans, details, complete, pendingBranches); - const groups = groupConversation(items, trace.spans); - const requestedId = requestedBranch ?? null; - const requestedGroups = conversationBranchGroups(groups, requestedId); - const requestedLimit = pages.get(requestedId) ?? CONVERSATION_PAGE_SIZE; - const sourceRemaining = hasMore || Boolean(trace.next_cursor && paging); - const requestedSourceRemaining = hasRemainingSource(requestedId, sourceRemaining, pendingBranches); - const needsMore = - requestedBranch !== undefined && requestedGroups.length < requestedLimit && requestedSourceRemaining; - const busy = loading || Boolean(paging?.loading); - const blocked = failed || Boolean(paging?.failed); - const loadTracePage = paging?.loadMore; - useEffect(() => { - if (!needsMore || busy || blocked) return; - if (hasMore) { - const controller = new AbortController(); - void loadMore(controller.signal); - return () => controller.abort(); - } - if (trace.next_cursor) loadTracePage?.(); - }, [needsMore, busy, blocked, hasMore, loadMore, trace.next_cursor, loadTracePage]); - const fillingPage = needsMore && sourceRemaining && !blocked; - const loadEntries = (branchId: string | null, shown: number) => { - setPages((current) => new Map([...current, [branchId, shown + CONVERSATION_PAGE_SIZE]])); - setRequestedBranch(branchId); - }; - const warnings = conversationWarnings(details, traceComplete); - const multipleAgents = new Set(items.map((item) => item.agentId).filter(Boolean)).size > 1; - const rootLimit = pages.get(null) ?? CONVERSATION_PAGE_SIZE; - const inlineErrorIds = new Set( - groups - .slice(0, rootLimit) - .flatMap((group) => (group.kind === "item" && group.item.showError ? [group.item.span.span_id] : [])), - ); - const rootErrors = trace.spans.filter((span) => { - const failedRoot = span.parent_span_id === null && span.status === "error" && span.type !== "tool"; - return failedRoot && !inlineErrorIds.has(span.span_id); - }); - return ( -
-
- {rootErrors.map((span) => ( - - ))} - {warnings.map((warning) => ( -

- {warning} -

- ))} - - -
-
- ); -} - -function ConversationFeedback({ - entries, - loading, - complete, - empty, - paging, -}: { - entries: ReturnType["entries"]; - loading: boolean; - complete: boolean; - empty: boolean; - paging?: ConversationTracePaging; -}) { - return ( - <> - {entries.map( - ({ query, span }) => - query.isError && ( -
- Could not load {span.name}. Retry this step to continue the conversation. - -
- ), - )} - {loading && ( -

- Loading conversation… -

- )} - {complete && empty &&

No conversation content recorded.

} - {paging?.failed && ( -

- Could not load more conversation entries. Use Retry or Refresh trace above to continue. -

- )} - - ); -} - -function ConversationGroups({ - groups, - multipleAgents, - onOpenStep, - branchId = null, - name, - pages, - sourceRemaining, - pendingBranches, - disabled, - onLoadEntries, - complete, -}: { - groups: ConversationGroup[]; - multipleAgents: boolean; - onOpenStep: (id: string) => void; - branchId?: string | null; - name?: string; - pages: ReadonlyMap; - sourceRemaining: boolean; - pendingBranches: ReadonlySet; - disabled: boolean; - onLoadEntries: (branchId: string | null, shown: number) => void; - complete: boolean; -}) { - const limit = pages.get(branchId) ?? CONVERSATION_PAGE_SIZE; - const visible = groups.slice(0, limit); - const branchRemaining = hasRemainingSource(branchId, sourceRemaining, pendingBranches); - const hasMore = visible.length < groups.length || branchRemaining; - const nextCount = branchRemaining - ? CONVERSATION_PAGE_SIZE - : Math.min(CONVERSATION_PAGE_SIZE, groups.length - visible.length); - return ( - <> - {visible.map((group) => - group.kind === "item" ? ( - - ) : ( -
- - Subagent: {group.name} - - {group.children.length} entries loaded - - -
- -
-
- ), - )} -
- - {complete && !hasMore && branchId === null ? "End of conversation" : `${visible.length} entries shown`} - - {hasMore && ( - - )} -
- - ); -} - -function ConversationStep({ - item, - multipleAgents, - onOpenStep, -}: { - item: ConversationItem; - multipleAgents: boolean; - onOpenStep: (id: string) => void; -}) { - return ( -
- {(multipleAgents || item.toolResult === undefined) && ( -
-
- {multipleAgents && ( - - {item.agentName} - - )} - {item.model && ( - - {item.model} - - )} - {fmtMs(item.time ?? item.span.start_offset_ms)} -
- {item.toolResult === undefined && ( - - )} -
- )} - {item.showError && } - {item.messages.map((message, index) => ( - - ))} - {item.toolResult !== undefined && } -
- ); -} - -function ConversationTool({ item, onOpenStep }: { item: ConversationItem; onOpenStep: (id: string) => void }) { - const [open, setOpen] = useState(item.span.status === "error"); - const failed = item.span.status === "error"; - const summary = toolSummary(item.toolCall?.args); - return ( -
-
- - -
- {open && ( -
- {item.toolCall && } - {item.span.error && item.span.error !== item.toolResult && ( -

{item.span.error}

- )} -
- Result - -
- {item.toolResult ? ( - - ) : ( -

No result content recorded.

- )} -
- )} -
- ); -} - -function ConversationMessage({ message }: { message: TraceMessage }) { - if (message.role === "system" && (message.content.startsWith("Agent ") || message.content === "Context compacted")) - return

{message.content}

; - if (message.role === "system") - return ( -
- Session context -
- -
-
- ); - if (message.role === "tool") return ; - return ( -
- {message.content && ( -
- -
- )} - {message.tool_calls?.map((call, index) => ( -
- {call.name} -
- -
-
- ))} -
- ); -} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.integration.test.tsx new file mode 100644 index 00000000000..fe56c045337 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.integration.test.tsx @@ -0,0 +1,253 @@ +import { screen, within } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import type { ComponentProps } from "react"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import { renderWithProviders, testQueryClient } from "../../../../../../tests/test-utils"; +import research from "../../__fixtures__/research_trace.json"; +import { useOpenTraceRouting } from "../../routing"; +import type { SpanDetail, Trace } from "../../types"; +import { RunView } from "../run/RunView"; + +vi.mock("../../../../networking", () => ({ + agentTraceCall: vi.fn(), + agentTraceSpanCall: vi.fn(), + getProxyBaseUrl: () => "http://proxy.test", +})); +import { agentTraceCall, agentTraceSpanCall } from "../../../../networking"; + +function RoutedRunView(props: Omit, "selection">) { + const { selection } = useOpenTraceRouting(); + return ; +} + +const root = { ...research.spans[0], span_id: "root", parent_span_id: null, type: "agent", duration_ms: 900 }; +const llm = { ...root, span_id: "llm", name: "chat", type: "llm", parent_span_id: "root", start_offset_ms: 10 }; +const tool = { ...root, span_id: "tool", name: "read_file", type: "tool", parent_span_id: "root", start_offset_ms: 50 }; +const trace = { ...research, spans: [root, llm, tool] } as Trace; +const details: Record = { + root: { + span_id: "root", + input: '[{"role":"user","content":"Read the release notes"}]', + output: '[{"role":"assistant","content":"The release is ready"}]', + attributes: {}, + }, + llm: { + span_id: "llm", + input: '[{"role":"user","content":"Read the release notes"}]', + output: JSON.stringify([ + { role: "assistant", content: "Checking the changelog", tool_calls: [{ name: "read_file", args: {} }] }, + ]), + attributes: {}, + }, + tool: { span_id: "tool", input: '{"path":"CHANGELOG.md"}', output: "All checks passed", attributes: {} }, +}; + +describe("TraceThread", () => { + beforeEach(() => { + testQueryClient.clear(); + vi.mocked(agentTraceCall).mockReset().mockResolvedValue(trace); + vi.mocked(agentTraceSpanCall) + .mockReset() + .mockImplementation(async (_token, _trace, id) => details[id]); + }); + + it("folds the agent's work under the prompt and shows only the final reply until expanded", async () => { + const user = userEvent.setup(); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + + expect(await within(thread).findByText("Read the release notes")).toBeVisible(); + expect(await within(thread).findByText("The release is ready")).toBeVisible(); + const worked = within(thread).getByRole("button", { name: /^Worked/, expanded: false }); + expect(within(worked).getByLabelText("1 model calls")).toBeVisible(); + expect(within(worked).getByLabelText("1 tool calls")).toBeVisible(); + expect(within(thread).queryByText("Checking the changelog")).not.toBeInTheDocument(); + + await user.click(worked); + expect(within(thread).getByText("Checking the changelog")).toBeVisible(); + await user.click(within(thread).getByRole("button", { name: "Inspect step read_file" })); + expect(screen.getByRole("tab", { name: "Steps", selected: true })).toBeVisible(); + expect(screen.getByRole("treeitem", { selected: true })).toHaveAttribute("data-row-id", "tool"); + }); + + it("retries a step that failed to load and then shows the reply", async () => { + const user = userEvent.setup(); + vi.mocked(agentTraceSpanCall) + .mockRejectedValueOnce(new Error("boom")) + .mockImplementation(async (_token, _trace, id) => details[id]); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + const alert = await within(thread).findByRole("alert"); + expect(alert).toHaveTextContent("Could not load some steps of this thread."); + await user.click(within(alert).getByRole("button", { name: "Retry" })); + expect(await within(thread).findByText("The release is ready")).toBeVisible(); + expect(within(thread).queryByRole("alert")).not.toBeInTheDocument(); + }); + + it("offers only Steps and Thread, and an old conversation link opens on Steps", async () => { + window.history.replaceState(null, "", "/?view=conversation"); + renderWithProviders( + , + ); + const views = await screen.findByRole("tablist", { name: "Trace view" }); + expect( + within(views) + .getAllByRole("tab") + .map((tab) => tab.textContent), + ).toEqual(["Steps", "Thread"]); + expect(within(views).getByRole("tab", { name: "Steps" })).toHaveAttribute("aria-selected", "true"); + window.history.replaceState(null, "", "/"); + }); + + it.each([false, true])("refreshes the reply in place and keeps it if the refresh fails (%s)", async (failed) => { + const user = userEvent.setup(); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + expect(await within(thread).findByText("The release is ready")).toBeVisible(); + vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { + if (failed) throw new Error("content refresh failed"); + return id === "root" + ? { ...details.root, output: '[{"role":"assistant","content":"Updated answer"}]' } + : details[id]; + }); + await user.click(screen.getByRole("button", { name: "Refresh run" })); + if (failed) { + expect(await within(thread).findByRole("button", { name: "Retry" })).toBeVisible(); + expect(within(thread).getByText("The release is ready")).toBeVisible(); + } else { + expect(await within(thread).findByText("Updated answer")).toBeVisible(); + expect(within(thread).queryByText("The release is ready")).not.toBeInTheDocument(); + } + }); + + it("keeps the whole thread on screen when a refresh of the run fails", async () => { + const user = userEvent.setup(); + const first = { ...trace, spans: [root, llm], next_cursor: "next-page" } as Trace; + const second = { ...trace, spans: [tool], next_cursor: null } as Trace; + vi.mocked(agentTraceCall).mockImplementation(async (_token, _trace, _ref, cursor) => (cursor ? second : first)); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + expect(await within(thread).findByText("End of thread")).toBeVisible(); + expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([null, "next-page"]); + vi.mocked(agentTraceCall).mockRejectedValue(new Error("refresh unavailable")); + await user.click(screen.getByRole("button", { name: "Refresh run" })); + expect(await screen.findByText(/Previously received steps are still shown/)).toBeVisible(); + expect(within(thread).getByText("Read the release notes")).toBeVisible(); + expect(within(thread).getByText("The release is ready")).toBeVisible(); + }); + + it("shows a failed shell command's output inside the Worked bar and in the step details", async () => { + const user = userEvent.setup(); + const command = "npm test -- checkout"; + const terminal = { ...tool, name: "terminal", status: "error", error: null }; + vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root, llm, terminal] } as Trace); + vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => + id === "tool" + ? { + span_id: "tool", + input: JSON.stringify({ command }), + output: JSON.stringify({ output: "FAIL checkout.test.ts", exit_code: 1 }), + input_ui: { kind: "fields", fields: [{ key: "command", value: command }] }, + output_ui: { + kind: "fields", + fields: [ + { key: "output", value: "FAIL checkout.test.ts" }, + { key: "exit_code", value: "1" }, + ], + }, + attributes: {}, + } + : details[id], + ); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + const worked = await within(thread).findByRole("button", { name: /^Worked/ }); + expect(within(worked).getByLabelText("A step failed")).toBeVisible(); + await user.click(worked); + expect(await within(thread).findByText(/FAIL checkout.test.ts/, { selector: "pre" })).toBeVisible(); + expect(within(thread).getByText("exit_code")).toBeVisible(); + await user.click(within(thread).getByRole("button", { name: "Inspect step terminal" })); + const pane = screen.getByRole("complementary", { name: "Span details" }); + expect(await within(pane).findByText(command, { selector: "pre" })).toBeVisible(); + }); + + it("shows an agent failure once even when the run never replied", async () => { + const user = userEvent.setup(); + const failedRoot = { ...root, status: "error", error: "Agent exceeded its execution limit" }; + vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [failedRoot, llm, tool] } as Trace); + vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => + id === "root" ? { ...details.root, output: "" } : details[id], + ); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + expect(await within(thread).findByText("End of thread")).toBeVisible(); + expect(within(thread).getAllByText("Agent exceeded its execution limit")).toHaveLength(1); + await user.click(within(thread).getByRole("button", { name: /^Worked/ })); + expect(within(thread).getAllByText("Agent exceeded its execution limit")).toHaveLength(1); + }); + + it("warns when a Claude Code trace recorded no assistant replies", async () => { + const user = userEvent.setup(); + vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root] } as Trace); + vi.mocked(agentTraceSpanCall).mockResolvedValue({ + ...details.root, + output: "", + attributes: { "span.type": "llm_request" }, + }); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + expect(await screen.findByText(/no recorded assistant replies/)).toBeVisible(); + expect(screen.getByText("Read the release notes")).toBeVisible(); + }); + + it("nests a failed subagent's work inside the Worked bar and shows its error once", async () => { + const user = userEvent.setup(); + const child = { + ...root, + span_id: "child", + parent_span_id: "root", + name: "Investigate release", + start_offset_ms: 20, + duration_ms: 10, + status: "error", + error: "Investigation timed out", + }; + const childTool = { ...tool, span_id: "child-tool", parent_span_id: "child", start_offset_ms: 22, duration_ms: 2 }; + vi.mocked(agentTraceCall).mockResolvedValue({ ...trace, spans: [root, llm, child, childTool] } as Trace); + vi.mocked(agentTraceSpanCall).mockImplementation(async (_token, _trace, id) => { + if (id === "child") return { ...details.root, span_id: id, input: "Investigate failed checks", output: "" }; + if (id === "child-tool") return { ...details.tool, span_id: id }; + return details[id]; + }); + renderWithProviders( + , + ); + await user.click(await screen.findByRole("tab", { name: "Thread" })); + const thread = await screen.findByRole("region", { name: "Trace thread" }); + await user.click(await within(thread).findByRole("button", { name: /^Worked/ })); + await user.click(within(thread).getByText("Subagent: Investigate release", { exact: true })); + expect(within(thread).getByText("Investigate failed checks")).toBeVisible(); + expect(within(thread).getAllByText("Investigation timed out")).toHaveLength(1); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.tsx new file mode 100644 index 00000000000..f82c50d731e --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceThread.tsx @@ -0,0 +1,258 @@ +"use client"; + +import { ChevronRight, MessageSquareText, TriangleAlert, Wrench } from "lucide-react"; +import { useEffect, useState } from "react"; + +import { Button } from "@/components/ui/button"; +import { cn } from "@/lib/cva.config"; + +import type { Trace } from "../../types"; +import { fmtMs } from "../../utils"; +import { ErrorBlock } from "../content/SpanError"; +import { Markdown } from "../content/Markdown"; +import { + buildConversation, + conversationWarnings, + groupConversation, + pendingConversationBranches, +} from "./conversation"; +import { ConversationMessage, ConversationStep, ConversationSteps } from "./ConversationParts"; +import { + buildThread, + replyErrorSpanIds, + threadDurationMs, + withoutErrorsOf, + type ThreadTurn, + type ThreadWork, +} from "./thread"; +import { useConversationDetails } from "./useConversationDetails"; + +export interface ConversationTracePaging { + loading: boolean; + failed: boolean; + loadMore: () => void; +} + +const AUTO_LOAD_STEPS = 200; + +interface TraceThreadProps { + trace: Trace; + accessToken: string; + onOpenStep: (id: string) => void; + paging?: ConversationTracePaging; +} + +export function TraceThread({ trace, accessToken, onOpenStep, paging }: TraceThreadProps) { + const { details, entries, complete, loading, failed, hasMore, loadMore } = useConversationDetails(trace, accessToken); + const pending = pendingConversationBranches(trace.spans, details, Boolean(trace.next_cursor)); + const groups = groupConversation(buildConversation(trace.spans, details, complete, pending), trace.spans); + const builtTurns = buildThread(groups); + const traceComplete = complete && !trace.next_cursor; + const shownErrors = replyErrorSpanIds(builtTurns); + const toolSpanIds = new Set(trace.spans.flatMap((span) => (span.type === "tool" ? [span.span_id] : []))); + const rootErrors = trace.spans.filter( + (span) => span.parent_span_id === null && span.status === "error" && !shownErrors.has(span.span_id), + ); + const rootErrorIds = new Set(rootErrors.map((span) => span.span_id)); + const turns = builtTurns.map((turn) => withoutErrorsOf(turn, rootErrorIds)); + const busy = loading || Boolean(paging?.loading); + const blocked = failed || Boolean(paging?.failed); + const sourceRemaining = hasMore || Boolean(trace.next_cursor && paging); + const canLoad = !busy && !blocked && sourceRemaining; + const autoLoad = canLoad && details.size < AUTO_LOAD_STEPS; + const loadTracePage = paging?.loadMore; + useEffect(() => { + if (!autoLoad) return; + if (!hasMore) { + loadTracePage?.(); + return; + } + const controller = new AbortController(); + void loadMore(controller.signal); + return () => controller.abort(); + }, [autoLoad, hasMore, loadMore, loadTracePage]); + const loadNext = () => { + if (hasMore) void loadMore(new AbortController().signal); + else loadTracePage?.(); + }; + return ( +
+
+ {rootErrors.map((span) => ( + + ))} + {conversationWarnings(details, traceComplete, toolSpanIds).map((warning) => ( +

+ {warning} +

+ ))} + {turns.map((turn) => ( + + ))} + entries.forEach(({ query }) => query.isError && void query.refetch())} + onLoadMore={canLoad && !autoLoad ? loadNext : undefined} + /> +
+
+ ); +} + +interface ThreadFooterProps { + turnCount: number; + busy: boolean; + stepsFailed: boolean; + traceFailed: boolean; + complete: boolean; + onRetry: () => void; + onLoadMore?: () => void; +} + +function ThreadFooter({ turnCount, busy, stepsFailed, traceFailed, complete, onRetry, onLoadMore }: ThreadFooterProps) { + return ( + <> + {busy && ( +

+ Loading thread… +

+ )} + {stepsFailed && ( +
+ Could not load some steps of this thread. + +
+ )} + {traceFailed && ( +

+ Could not load more of this trace. Use Retry or Refresh trace above. +

+ )} + {complete && !turnCount &&

No conversation content recorded.

} +
+ {complete ? "End of thread" : `${turnCount} ${turnCount === 1 ? "turn" : "turns"} loaded`} + {onLoadMore && ( + + )} +
+ + ); +} + +function ThreadTurnView({ turn, onOpenStep }: { turn: ThreadTurn; onOpenStep: (id: string) => void }) { + return ( +
+ {turn.context.length > 0 && } + {turn.prompt.map((message, index) => ( + + ))} + {turn.work.length > 0 && } + {turn.replyItem?.showError && } + {turn.reply && ( +
+
+ +
+ +
+ )} +
+ ); +} + +function PromptContext({ count, turn }: { count: number; turn: ThreadTurn }) { + return ( +
+ + Prompt context + + {count} + + +
+ {turn.context.map((message, index) => ( + + ))} +
+
+ ); +} + +function WorkedBar({ turn, onOpenStep }: { turn: ThreadTurn; onOpenStep: (id: string) => void }) { + const [open, setOpen] = useState(false); + const duration = threadDurationMs(turn); + return ( +
+ + {open && ( +
+ {turn.work.map((work) => ( + + ))} +
+ )} +
+ ); +} + +function Count({ icon: Icon, value, label }: { icon: typeof Wrench; value: number; label: string }) { + return ( + + + {value} + + ); +} + +function WorkItem({ work, onOpenStep }: { work: ThreadWork; onOpenStep: (id: string) => void }) { + if (work.kind === "step") return ; + return ( +
+ Subagent: {work.name} +
+ +
+
+ ); +} + +function ReplyMeta({ turn, onOpenStep }: { turn: ThreadTurn; onOpenStep: (id: string) => void }) { + const item = turn.replyItem; + if (!item) return null; + const tokens = item.span.input_tokens + item.span.output_tokens; + return ( +
+ {item.model && {item.model}} + {fmtMs(item.span.duration_ms)} + {tokens > 0 && {tokens.toLocaleString()} tok} + +
+ ); +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts index e4d47a94f90..53a2ee1b7f4 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts @@ -683,4 +683,20 @@ describe("coding sessions", () => { details.set("reply", { ...detail("reply", [], "Hello"), attributes: { "event.name": "assistant_response" } }); expect(conversationWarnings(details, true)).toEqual([]); }); + + it("still warns about missing replies when only a tool recorded output", () => { + const details = new Map([ + ["llm", { ...detail("llm", [], []), output: "", attributes: { "span.type": "llm_request" } }], + ["read", { ...detail("read", {}, "file contents"), output: "file contents" }], + ]); + expect(conversationWarnings(details, true, new Set(["read"]))).toHaveLength(1); + }); + + it("does not warn about missing replies when another model call recorded the answer", () => { + const details = new Map([ + ["llm", { ...detail("llm", [], []), output: "", attributes: { "span.type": "llm_request" } }], + ["answer", detail("answer", [], [answer])], + ]); + expect(conversationWarnings(details, true)).toEqual([]); + }); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts index 573f9eb9f9b..f26c1996b14 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts @@ -451,23 +451,35 @@ export function groupConversation(items: readonly ConversationItem[], spans: rea return build(); } -export function conversationWarnings(details: ReadonlyMap, complete: boolean): string[] { +export function conversationWarnings( + details: ReadonlyMap, + complete: boolean, + toolSpanIds: ReadonlySet = new Set(), +): string[] { const warnings = [ ...claudeCaptureWarnings(details), ...[...details.values()].flatMap((detail) => detail.attributes["lens.capture.warning"] ? [detail.attributes["lens.capture.warning"]] : [], ), ]; - if ( - complete && - ![...details.values()].some((detail) => detail.attributes["event.name"] === "assistant_response") && - [...details.values()].some( - (detail) => - detail.attributes["span.type"] === "llm_request" && - !detail.output && - !messages(detail.output, detail.output_ui, "assistant").length, - ) - ) { + const hasAssistantText = [...details.values()].some( + (detail) => + !toolSpanIds.has(detail.span_id) && + messages(detail.output, detail.output_ui, "assistant").some( + (message) => message.role === "assistant" && message.content.trim(), + ), + ); + const hasReplyEvent = [...details.values()].some( + (detail) => detail.attributes["event.name"] === "assistant_response", + ); + const hasEmptyRequest = [...details.values()].some( + (detail) => + detail.attributes["span.type"] === "llm_request" && + !detail.output && + !messages(detail.output, detail.output_ui, "assistant").length, + ); + const missingReplies = !hasAssistantText && !hasReplyEvent && hasEmptyRequest; + if (complete && missingReplies) { warnings.push( "This Claude Code trace has no recorded assistant replies. Enable assistant response logs for future sessions.", ); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.test.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.test.ts new file mode 100644 index 00000000000..a08b81f5815 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.test.ts @@ -0,0 +1,194 @@ +import { describe, expect, it } from "vitest"; +import research from "../../__fixtures__/research_trace.json"; +import type { Span, TraceMessage } from "../../types"; +import type { ConversationGroup, ConversationItem } from "./conversation"; +import { buildThread, replyErrorSpanIds, threadDurationMs, withoutErrorsOf } from "./thread"; + +const base = research.spans[0] as Span; +const span = (id: string, type: Span["type"], start: number, timing: Partial = {}): Span => ({ + ...base, + span_id: id, + type, + start_offset_ms: start, + duration_ms: 100, + status: "ok", + ...timing, +}); +const item = (id: string, type: Span["type"], start: number, messages: TraceMessage[]): ConversationGroup => ({ + kind: "item", + item: { id, span: span(id, type, start), messages } as ConversationItem, +}); +const tool = (id: string, start: number, status: Span["status"] = "ok"): ConversationGroup => ({ + kind: "item", + item: { id, span: span(id, "tool", start, { duration_ms: 50, status }), messages: [], toolResult: "ok" }, +}); + +const ask = (content: string): TraceMessage => ({ role: "user", content }); +const say = (content: string): TraceMessage => ({ role: "assistant", content }); +const plan = (content: string): TraceMessage => ({ + role: "assistant", + content, + tool_calls: [{ name: "search", args: {} }], +}); + +describe("buildThread", () => { + it("folds the calls between a prompt and its final answer into one turn", () => { + const turns = buildThread([ + item("root", "agent", 0, [ask("fix the bug")]), + item("llm1", "llm", 10, [plan("looking")]), + tool("t1", 200), + item("llm2", "llm", 300, [say("Fixed it")]), + ]); + expect(turns).toHaveLength(1); + expect(turns[0].prompt.map((m) => m.content)).toEqual(["fix the bug"]); + expect(turns[0].reply?.content).toBe("Fixed it"); + expect(turns[0].work.map((w) => (w.kind === "step" ? w.item.id : w.id))).toEqual(["llm1", "t1"]); + expect(turns[0].llmCalls).toBe(1); + expect(turns[0].toolCalls).toBe(1); + expect(threadDurationMs(turns[0])).toBe(400); + }); + + it("starts a new turn at every user prompt", () => { + const turns = buildThread([ + item("u1", "agent", 0, [ask("first")]), + item("a1", "llm", 10, [say("one")]), + item("u2", "agent", 500, [ask("second")]), + item("a2", "llm", 510, [say("two")]), + ]); + expect(turns.map((t) => [t.prompt[0]?.content, t.reply?.content])).toEqual([ + ["first", "one"], + ["second", "two"], + ]); + }); + + it("keeps a reply's tool calls out of the reply and in the work", () => { + const turns = buildThread([ + item("u", "agent", 0, [ask("go")]), + item("llm", "llm", 10, [plan("calling tools"), say("done")]), + ]); + expect(turns[0].reply?.content).toBe("done"); + const leftover = turns[0].work.at(-1); + expect(leftover?.kind === "step" && leftover.item.messages.map((m) => m.content)).toEqual(["calling tools"]); + }); + + it("has no reply when the agent only made tool calls", () => { + const turns = buildThread([item("u", "agent", 0, [ask("go")]), item("llm", "llm", 10, [plan("x")]), tool("t", 20)]); + expect(turns[0].reply).toBeNull(); + expect(turns[0].work).toHaveLength(2); + }); + + it("separates system context from the prompt", () => { + const turns = buildThread([item("u", "agent", 0, [{ role: "system", content: "You are helpful" }, ask("hi")])]); + expect(turns[0].context.map((m) => m.content)).toEqual(["You are helpful"]); + expect(turns[0].prompt.map((m) => m.content)).toEqual(["hi"]); + }); + + it("counts subagent work and marks the turn failed when a nested step failed", () => { + const turns = buildThread([ + item("u", "agent", 0, [ask("research")]), + { + kind: "branch", + id: "sub", + name: "Explore", + children: [item("s1", "llm", 50, [plan("searching")]), tool("s2", 80, "error")], + }, + item("a", "llm", 900, [say("summary")]), + ]); + expect(turns[0].work[0]).toMatchObject({ kind: "subagent", name: "Explore" }); + expect(turns[0].llmCalls).toBe(1); + expect(turns[0].toolCalls).toBe(1); + expect(turns[0].failed).toBe(true); + }); + + it("keeps work before the first prompt as its own turn", () => { + const turns = buildThread([ + tool("warmup", 0), + item("u", "agent", 10, [ask("hi")]), + item("a", "llm", 20, [say("hey")]), + ]); + expect(turns).toHaveLength(2); + expect(turns[0].prompt).toEqual([]); + expect(turns[1].reply?.content).toBe("hey"); + }); + + it("treats an LLM call that re-sends the same prompt as part of the turn, not a new one", () => { + const turns = buildThread([ + item("root", "agent", 0, [ask("What is an agent trace?")]), + item("llm1", "llm", 10, [{ role: "system", content: "Be brief" }, ask("What is an agent trace?"), plan("look")]), + tool("t1", 100), + item("llm2", "llm", 200, [say("A recording of one agent run")]), + ]); + expect(turns).toHaveLength(1); + expect(turns[0].prompt.map((m) => m.content)).toEqual(["What is an agent trace?"]); + expect(turns[0].context.map((m) => m.content)).toEqual(["Be brief"]); + expect(turns[0].llmCalls).toBe(1); + expect(turns[0].reply?.content).toBe("A recording of one agent run"); + }); + + it("starts a new turn when the user asks the same thing again after an answer", () => { + const turns = buildThread([ + item("u1", "agent", 0, [ask("again?")]), + item("a1", "llm", 10, [say("yes")]), + item("u2", "agent", 100, [ask("again?")]), + item("a2", "llm", 110, [say("still yes")]), + ]); + expect(turns.map((t) => t.reply?.content)).toEqual(["yes", "still yes"]); + }); + + it("keeps a failure that happened after the reply in the work and marks the turn failed", () => { + const turns = buildThread([ + item("u", "agent", 0, [ask("deploy")]), + item("llm", "llm", 10, [say("Deployed")]), + tool("cleanup", 200, "error"), + ]); + expect(turns[0].reply?.content).toBe("Deployed"); + expect(turns[0].work.map((w) => (w.kind === "step" ? w.item.id : w.id))).toEqual(["cleanup"]); + expect(turns[0].failed).toBe(true); + }); + + it("uses a subagent's answer as the reply when the root only forwarded it", () => { + const turns = buildThread([ + item("u", "agent", 0, [ask("research")]), + { + kind: "branch", + id: "sub", + name: "Explore", + children: [item("s1", "llm", 50, [plan("searching")]), item("s2", "llm", 90, [say("Found the answer")])], + }, + ]); + expect(turns[0].reply?.content).toBe("Found the answer"); + expect(turns[0].replyItem?.id).toBe("s2"); + expect(turns[0].work[0]).toMatchObject({ kind: "subagent", name: "Explore" }); + }); + + it("picks the subagent answer that finished last when branches run in parallel", () => { + const branch = (id: string, start: number, answer: string): ConversationGroup => ({ + kind: "branch", + id, + name: id, + children: [item(`${id}-a`, "llm", start, [say(answer)])], + }); + const turns = buildThread([ + item("u", "agent", 0, [ask("compare")]), + branch("slow", 500, "Slow finished last"), + branch("fast", 100, "Fast finished first"), + ]); + expect(turns[0].reply?.content).toBe("Slow finished last"); + }); + + it("drops a run-level error from the Worked bar so it shows once at the top", () => { + const failed: ConversationGroup = { + kind: "item", + item: { id: "root-out", span: span("root", "agent", 0, { status: "error" }), messages: [], showError: true }, + }; + const [turn] = buildThread([item("u", "agent", 0, [ask("go")]), tool("t", 10), failed]); + expect(turn.work.map((w) => (w.kind === "step" ? w.item.id : w.id))).toEqual(["t", "root-out"]); + const cleaned = withoutErrorsOf(turn, new Set(["root"])); + expect(cleaned.work.map((w) => (w.kind === "step" ? w.item.id : w.id))).toEqual(["t"]); + expect(replyErrorSpanIds([turn]).size).toBe(0); + }); + + it("returns no turns for an empty conversation", () => { + expect(buildThread([])).toEqual([]); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.ts new file mode 100644 index 00000000000..0dcd9896aa1 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/thread.ts @@ -0,0 +1,173 @@ +import type { TraceMessage } from "../../types"; +import type { ConversationGroup, ConversationItem } from "./conversation"; + +export type ThreadWork = + | { kind: "step"; item: ConversationItem } + | { kind: "subagent"; id: string; name: string; groups: readonly ConversationGroup[] }; + +export interface ThreadTurn { + id: string; + prompt: readonly TraceMessage[]; + context: readonly TraceMessage[]; + work: readonly ThreadWork[]; + reply: TraceMessage | null; + replyItem: ConversationItem | null; + startMs: number; + endMs: number; + llmCalls: number; + toolCalls: number; + failed: boolean; +} + +interface Piece { + work: ThreadWork; + start: number; + end: number; + messages: readonly TraceMessage[]; +} + +const isPrompt = (message: TraceMessage): boolean => message.role === "user"; +const isContext = (message: TraceMessage): boolean => message.role === "system"; +const isReply = (message: TraceMessage): boolean => + message.role === "assistant" && Boolean(message.content.trim()) && !message.tool_calls?.length; + +function itemSpanMs(item: ConversationItem): { start: number; end: number } { + const start = item.time ?? item.span.start_offset_ms; + return { start, end: Math.max(start, item.span.start_offset_ms + item.span.duration_ms) }; +} + +function flatItems(groups: readonly ConversationGroup[]): ConversationItem[] { + return groups.flatMap((group) => (group.kind === "item" ? [group.item] : flatItems(group.children))); +} + +function piece(group: ConversationGroup): Piece { + if (group.kind === "item") + return { work: { kind: "step", item: group.item }, ...itemSpanMs(group.item), messages: group.item.messages }; + const items = flatItems(group.children); + const times = items.map(itemSpanMs); + return { + work: { kind: "subagent", id: group.id, name: group.name, groups: group.children }, + start: Math.min(...times.map((t) => t.start), Infinity), + end: Math.max(...times.map((t) => t.end), -Infinity), + messages: [], + }; +} + +function countWork(work: readonly ThreadWork[]): { llmCalls: number; toolCalls: number; failed: boolean } { + const items = work.flatMap((w) => (w.kind === "step" ? [w.item] : flatItems(w.groups))); + return { + llmCalls: items.filter((item) => item.span.type === "llm").length, + toolCalls: items.filter((item) => item.toolResult !== undefined).length, + failed: items.some((item) => item.span.status === "error"), + }; +} + +function stripMessages(item: ConversationItem, keep: (message: TraceMessage) => boolean): ConversationItem | null { + const messages = item.messages.filter(keep); + const empty = !messages.length && item.toolResult === undefined && !item.showError; + return empty ? null : { ...item, messages }; +} + +interface TurnHead { + id: string; + prompt: readonly TraceMessage[]; + context: readonly TraceMessage[]; + promptStart: number; +} + +function lastReply(pieces: readonly Piece[]): { item: ConversationItem; message: TraceMessage } | null { + const direct = pieces.flatMap((p) => (p.work.kind === "step" ? [p.work.item] : [])); + const nested = pieces.flatMap((p) => (p.work.kind === "subagent" ? flatItems(p.work.groups) : [])); + const latest = (items: readonly ConversationItem[]) => + items + .filter((i) => i.messages.some(isReply)) + .reduce< + ConversationItem | undefined + >((best, i) => (!best || itemSpanMs(i).end >= itemSpanMs(best).end ? i : best), undefined); + const item = direct.findLast((i) => i.messages.some(isReply)) ?? latest(nested); + const message = item?.messages.findLast(isReply); + return item && message ? { item, message } : null; +} + +function closeTurn({ id, prompt, context, promptStart }: TurnHead, pieces: readonly Piece[]): ThreadTurn { + const found = lastReply(pieces); + const work = pieces.flatMap((p): ThreadWork[] => { + if (p.work.kind !== "step" || p.work.item !== found?.item) return [p.work]; + const leftover = stripMessages(p.work.item, (message) => message !== found.message); + return leftover ? [{ kind: "step", item: leftover }] : []; + }); + const timed = pieces.filter((p) => Number.isFinite(p.start)); + return { + id, + prompt, + context, + work, + reply: found?.message ?? null, + replyItem: found?.item ?? null, + startMs: Math.min(promptStart, ...timed.map((p) => p.start)), + endMs: Math.max(...timed.map((p) => p.end), -Infinity), + ...countWork(work), + }; +} + +export function buildThread(groups: readonly ConversationGroup[]): ThreadTurn[] { + const pieces = groups.map(piece); + const prompts = pieces.map((p) => + p.messages + .filter(isPrompt) + .map((m) => m.content) + .join("\n"), + ); + const answeredBefore = pieces.map((_, index) => + pieces.slice(0, index).findLastIndex((p) => p.messages.some(isReply)), + ); + const repeatsPrompt = (index: number): boolean => { + const previous = prompts.slice(0, index).findLastIndex(Boolean); + return previous >= 0 && prompts[previous] === prompts[index] && answeredBefore[index] < previous; + }; + const promptStarts = prompts.flatMap((text, index) => (text && !repeatsPrompt(index) ? [index] : [])); + const starts = promptStarts[0] === 0 ? promptStarts : [0, ...promptStarts]; + const repeats = new Set(prompts.flatMap((text, index) => (text && repeatsPrompt(index) ? [index] : []))); + const stripRepeat = (p: Piece, index: number): Piece[] => { + if (!repeats.has(index) || p.work.kind !== "step") return [p]; + const rest = stripMessages(p.work.item, (m) => !isPrompt(m) && !isContext(m)); + return rest ? [{ ...p, work: { kind: "step", item: rest }, messages: rest.messages }] : []; + }; + return starts.flatMap((start, index) => { + const end = starts[index + 1] ?? pieces.length; + const head = pieces[start]; + if (!head) return []; + const headItem = head.work.kind === "step" ? head.work.item : null; + const prompt = head.messages.filter(isPrompt); + const repeatContext = pieces + .slice(start + 1, end) + .flatMap((p, offset) => (repeats.has(start + 1 + offset) ? p.messages.filter(isContext) : [])); + const context = [...head.messages.filter(isContext), ...repeatContext]; + const rest = headItem ? stripMessages(headItem, (m) => !isPrompt(m) && !isContext(m)) : null; + const body = [ + ...(head.work.kind === "subagent" ? [head] : []), + ...(rest ? [{ ...head, work: { kind: "step" as const, item: rest }, messages: rest.messages }] : []), + ...pieces.slice(start + 1, end).flatMap((p, offset) => stripRepeat(p, start + 1 + offset)), + ]; + const promptStart = prompt.length ? head.start : Infinity; + const turnHead: TurnHead = { id: headItem?.id ?? `turn-${start}`, prompt, context, promptStart }; + return [closeTurn(turnHead, body)]; + }); +} + +export function replyErrorSpanIds(turns: readonly ThreadTurn[]): ReadonlySet { + return new Set(turns.flatMap((turn) => (turn.replyItem?.showError ? [turn.replyItem.span.span_id] : []))); +} + +export function withoutErrorsOf(turn: ThreadTurn, spanIds: ReadonlySet): ThreadTurn { + const work = turn.work.flatMap((w): ThreadWork[] => { + if (w.kind !== "step" || !w.item.showError || !spanIds.has(w.item.span.span_id)) return [w]; + const item = { ...w.item, showError: false }; + return item.messages.length || item.toolResult !== undefined ? [{ kind: "step", item }] : []; + }); + return { ...turn, work }; +} + +export function threadDurationMs(turn: ThreadTurn): number | null { + return Number.isFinite(turn.startMs) && Number.isFinite(turn.endMs) ? turn.endMs - turn.startMs : null; +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunBody.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunBody.tsx index b7019d07fae..25455f19f4a 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunBody.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunBody.tsx @@ -10,7 +10,7 @@ import { TabsContent } from "@/components/ui/tabs"; import type { RunSelection } from "../../routing"; import type { Trace } from "../../types"; -import { TraceConversation, type ConversationTracePaging } from "../conversation/TraceConversation"; +import { TraceThread, type ConversationTracePaging } from "../conversation/TraceThread"; import { DetailPane } from "../span/DetailPane"; import { SpanTree } from "../tree/SpanTree"; import type { TreeLayout } from "../tree/TreeRows"; @@ -53,18 +53,15 @@ export function RunBody({ trace, accessToken, selection, embedded, stale, conver useShortcut("right", () => tree.fold(true), { ...pane, description: "fold" }); useShortcut("escape", () => setDetailOpen(false), { ...pane, enabled: active && detailOpen, description: "close" }); - if (view === "conversation") + const openStep = (id: string) => { + select(id); + setView("steps"); + }; + + if (view === "thread") return ( - - { - select(id); - setView("steps"); - }} - /> + + ); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx index 7f6b2d0bd64..8c9587127e5 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx @@ -1,6 +1,6 @@ "use client"; -import { ArrowLeft, Check, Copy, Link, RefreshCw } from "lucide-react"; +import { ArrowLeft, Check, Copy, Link, ListTree, MessagesSquare, RefreshCw } from "lucide-react"; import { useState } from "react"; import { useTimeout } from "usehooks-ts"; @@ -122,11 +122,13 @@ export function RunHeader({
- + + Steps - - Conversation + + + Thread
diff --git a/ui/litellm-dashboard/src/components/lens/traces/routing.ts b/ui/litellm-dashboard/src/components/lens/traces/routing.ts index 3788dc2dff7..6715664da62 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/routing.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/routing.ts @@ -4,7 +4,7 @@ import { useCallback, useState } from "react"; import { TIME_RANGE_PARSERS } from "@/components/shared/timeRange/routing"; import type { TraceSummary } from "./types"; -export const TRACE_VIEWS = ["steps", "conversation"] as const; +export const TRACE_VIEWS = ["steps", "thread"] as const; export type TraceView = (typeof TRACE_VIEWS)[number]; export const SPAN_TABS = ["content", "request", "attributes"] as const;