fix(lens): group live conclusions by check with short labels

There is now one group per check_id: issue traces are the main count and pattern traces a secondary note, so there are no duplicate red and grey cards. A long instruction falls back to the humanized check id. Adds briefReasoning and traceRows for the simplified trace list, and drops helpers nothing uses.
This commit is contained in:
Ishaan Jaff 2026-10-03 17:46:52 -07:00
parent 10260da169
commit 55d9425a82
No known key found for this signature in database
6 changed files with 91 additions and 596 deletions

View file

@ -1,178 +0,0 @@
import { Wrench } from "lucide-react";
import { cn } from "@/lib/cva.config";
import { outcome, type Outcome, type Phase } from "../../model/live";
import { timeline, type TimelineItem } from "../../model/spanPreview";
import type { Review } from "../../model/types";
import { ModelName } from "./LiveStrip";
const RED = "text-[#e5484d]";
const CHIP: Record<Outcome, { label: string; className: string }> = {
issue: { label: "Issue", className: "bg-[#e5484d]/10 text-[#e5484d] ring-1 ring-inset ring-[#e5484d]/30" },
clear: { label: "Looks fine", className: "bg-muted text-foreground" },
unknown: { label: "Not enough evidence", className: "bg-muted/60 text-muted-foreground" },
};
const PANE_LABEL = "text-[11px] font-medium text-muted-foreground";
function VerdictChip({ result, decided }: { result: Outcome; decided: boolean }) {
if (!decided) {
return <span className="rounded-full bg-muted/60 px-2 py-0.5 text-[11px] text-muted-foreground">Reading…</span>;
}
const chip = CHIP[result];
return (
<span
data-testid="verdict-chip"
className={cn(
"rounded-full px-2 py-0.5 text-[11px] font-medium motion-safe:animate-in motion-safe:zoom-in-95 motion-safe:fade-in",
chip.className,
)}
>
{chip.label}
</span>
);
}
function Bubble({ who, text, tone }: { who: string; text: string; tone: "ask" | "reply" }) {
return (
<div className={cn("flex flex-col gap-1", tone === "reply" && "items-end")}>
<span className="text-[11px] text-muted-foreground">{who}</span>
<p
className={cn(
"max-w-[92%] rounded-2xl px-3 py-2 text-[13px] leading-relaxed break-words whitespace-pre-wrap text-foreground",
tone === "ask" ? "rounded-tl-sm bg-muted/70" : "rounded-tr-sm border bg-background",
)}
>
{text}
</p>
</div>
);
}
function ToolRow({ item }: { item: Extract<TimelineItem, { kind: "tool" }> }) {
return (
<div className="flex gap-2 text-[12px]">
<Wrench aria-hidden="true" className={cn("mt-0.5 size-3.5 shrink-0 text-muted-foreground", item.error && RED)} />
<div className="min-w-0 flex-1">
<p className="flex min-w-0 gap-1.5">
<span className="shrink-0 font-medium text-foreground">{item.name}</span>
<span className="truncate text-muted-foreground">{item.args}</span>
</p>
{item.result && (
<p className={cn("line-clamp-2 break-words", item.error ? RED : "text-muted-foreground")}>
<span aria-hidden="true">↳ </span>
{item.result}
</p>
)}
</div>
</div>
);
}
function Step({ item }: { item: TimelineItem }) {
switch (item.kind) {
case "ask":
return <Bubble who="User asked" text={item.text} tone="ask" />;
case "reply":
return <Bubble who="Agent replied" text={item.text} tone="reply" />;
case "tool":
return <ToolRow item={item} />;
case "failure":
return <p className={cn("text-[12px]", RED)}>{item.text}</p>;
case "note":
return (
<p className="line-clamp-2 text-[12px] text-muted-foreground">
<span className="mr-1.5 font-medium text-foreground">{item.label}</span>
{item.text}
</p>
);
}
}
function WhatHappened({ review, items, phase }: { review: Review; items: readonly TimelineItem[]; phase: Phase }) {
if (!items.length) return <p className="text-[12px] text-muted-foreground">No trace parts were shown to the model</p>;
return (
<ol aria-label="What happened" className="flex flex-col gap-1">
{items.map((item, index) => {
const reading = phase.span === index;
const cited = phase.verdict && !!review.spans[item.span]?.cited;
return (
<li
key={`${item.kind}-${item.span}-${index}`}
data-state={reading ? "reading" : cited ? "cited" : "idle"}
className={cn(
"rounded-lg px-2.5 py-2 transition-colors duration-150 motion-reduce:transition-none",
reading && "bg-trace-row-hover ring-1 ring-inset ring-foreground/15",
cited && "ring-[1.5px] ring-inset ring-[#e5484d]",
)}
>
<Step item={item} />
</li>
);
})}
</ol>
);
}
export function ReadingPanel({ review, phase }: { review: Review; phase: Phase }) {
const result = outcome(review);
const items = timeline(review.spans);
const typing = phase.typed < review.reasoning.length;
const verdicts = review.verdicts.length
? review.verdicts
: [{ check_id: "", kind: "pattern" as const, summary: review.cannot_assess ? "Not enough evidence to judge" : "No issue observed" }];
return (
<div className="flex flex-col gap-5">
<header className="flex items-start justify-between gap-3">
<div className="min-w-0">
<h3 className="truncate text-[14px] font-semibold text-foreground">{review.agent || review.name}</h3>
<p className="truncate font-mono text-[11px] text-muted-foreground">trace {review.trace_id}</p>
</div>
<VerdictChip result={result} decided={phase.verdict} />
</header>
<section className="flex flex-col gap-2">
<h4 className={PANE_LABEL}>What happened</h4>
<WhatHappened review={review} items={items} phase={phase} />
</section>
<section aria-label="Reasoning" className="flex flex-col gap-2">
<h4 className={cn(PANE_LABEL, "flex items-center gap-2")}>
<span>Lens&apos;s reasoning</span>
<ModelName model={review.model} />
</h4>
<p className="min-h-10 text-[13px] leading-relaxed whitespace-pre-wrap text-foreground">
{review.reasoning ? review.reasoning.slice(0, phase.typed) : !typing && "No reasoning was returned."}
{typing && <span aria-hidden="true" className="ml-px inline-block h-3.5 w-px translate-y-0.5 bg-foreground" />}
</p>
</section>
<ul
role="status"
aria-label="Verdict"
data-outcome={result}
className={cn(
"flex flex-col gap-1.5 transition-opacity duration-150 motion-reduce:transition-none",
phase.verdict ? "opacity-100" : "opacity-0",
)}
>
{phase.verdict &&
verdicts.map((verdict, index) => (
<li
key={`${verdict.check_id}-${index}`}
className={cn(
"flex gap-2 text-[13px] leading-snug",
verdict.kind === "issue" ? RED : result === "unknown" ? "text-muted-foreground" : "text-foreground",
)}
>
<span
aria-hidden="true"
className={cn(
"mt-1.5 size-1.5 shrink-0 rounded-full",
verdict.kind === "issue" ? "bg-[#e5484d]" : "bg-muted-foreground/40",
)}
/>
<span className="min-w-0">{verdict.summary}</span>
</li>
))}
</ul>
</div>
);
}

View file

@ -1,66 +0,0 @@
import { Loader2 } from "lucide-react";
import { agoLabel } from "@/components/view_logs/TraceView/lensField";
import { useNow } from "@/hooks/useNow";
import { cn } from "@/lib/cva.config";
import { inGroup, outcome, queueRows, reviewKey, shortVerdict, type Playback } from "../../model/live";
import type { Review } from "../../model/types";
const LIMIT = 60;
export function ReviewQueue({
playback,
live,
focused,
group,
onPick,
}: {
playback: Pick<Playback, "played" | "current">;
live: boolean;
focused: Review | null;
group: string | null;
onPick: (review: Review) => void;
}) {
const now = useNow(5000);
const rows = queueRows(playback, LIMIT).filter((review) => inGroup(review, group));
if (!rows.length) return <p className="py-2 text-[12px] text-muted-foreground">No traces in this group yet.</p>;
return (
<ol aria-label="Reviewed traces" className="flex flex-col">
{rows.map((review) => {
const reading = live && review === playback.current;
const result = outcome(review);
const selected = review === focused;
return (
<li key={reviewKey(review)} className={reading ? "motion-safe:animate-in motion-safe:fade-in" : ""}>
<button
type="button"
aria-current={selected ? "true" : undefined}
onClick={() => onPick(review)}
className={cn(
"grid w-full grid-cols-[0.75rem_minmax(0,7.5rem)_minmax(0,1fr)_auto] items-center gap-2 rounded-md px-2 py-1.5 text-left text-[12px] hover:bg-muted",
selected && "bg-background ring-[1.5px] ring-inset ring-foreground",
)}
>
{reading ? (
<Loader2 aria-label="Reviewing" className="size-3 text-muted-foreground motion-safe:animate-spin" />
) : (
<span
aria-label={result}
className={cn("size-1.5 rounded-full", result === "issue" ? "bg-[#e5484d]" : "bg-muted-foreground/40")}
/>
)}
<span className="truncate text-foreground">{review.agent || review.name}</span>
<span className={cn("truncate", result === "issue" ? "text-[#e5484d]" : "text-muted-foreground")}>
{reading ? "reading…" : shortVerdict(review)}
</span>
<span className="text-[11px] tabular-nums text-muted-foreground">
{agoLabel(Date.parse(review.at), now)}
</span>
</button>
</li>
);
})}
</ol>
);
}

View file

@ -1,9 +1,11 @@
import { describe, expect, it } from "vitest";
import {
analysisModel,
briefReasoning,
checkLabel,
conclusions,
traceRows,
decidedReviews,
focusedReview,
grownGroups,
inGroup,
share,
@ -12,7 +14,6 @@ import {
EMPTY_FEED,
shortVerdict,
stripState,
tickerLine,
liveJob,
liveStats,
outcome,
@ -27,7 +28,6 @@ import {
startPlayback,
stepDuration,
tokenLabel,
verdictLine,
} from "./live";
import type { Job, Review } from "./types";
@ -83,40 +83,41 @@ describe("review outcome", () => {
expect(outcome(review("a", { cannot_assess: true, verdicts: [issue("i")] }))).toBe("unknown");
});
it("names the agent and short trace id in the ticker, falling back to the run name", () => {
expect(tickerLine(review("a", { trace_id: "a91f3c02deadbeef" }))).toBe("reading support-bot · a91f3c02");
expect(tickerLine(review("a", { agent: "", name: "refund", trace_id: "7d21" }))).toBe("reading refund · 7d21");
});
it("leads the verdict line with the issue summary over patterns", () => {
expect(verdictLine(review("a", { verdicts: [pattern("p", "fine"), issue("i", "made it up")] }))).toBe(
"made it up",
);
expect(verdictLine(review("a"))).toBe("No issue observed");
expect(verdictLine(review("a", { cannot_assess: true }))).toBe("Not enough evidence to judge");
it("keeps the first few sentences of the reasoning", () => {
expect(briefReasoning("One. Two? Three! Four. Five.")).toBe("One. Two? Three!");
expect(briefReasoning("Saw 1.5 percent drop. Fine")).toBe("Saw 1.5 percent drop. Fine");
expect(briefReasoning(" ")).toBe("");
});
});
describe("conclusions", () => {
it("counts traces per check and kind, ranking issues before patterns, then by count", () => {
it("makes one group per check, counting issue traces and noting pattern traces", () => {
const reviews = [
review("a", { verdicts: [pattern("calm"), issue("invented", "first")] }),
review("b", { verdicts: [pattern("calm"), pattern("calm")] }),
review("c", { verdicts: [issue("unhappy_user")] }),
review("d", { verdicts: [issue("invented", "latest"), pattern("invented")] }),
review("d", { verdicts: [issue("invented", "latest"), pattern("invented", "fine")] }),
review("e", { verdicts: [pattern("invented", "fine")] }),
];
const result = conclusions(reviews, [{ id: "invented", instruction: "Invents answers", enabled: true }]);
expect(result.map((c) => [c.checkId, c.count, c.issue])).toEqual([
["invented", 2, true],
["unhappy_user", 1, true],
["calm", 2, false],
["invented", 1, false],
expect(result.map((c) => [c.checkId, c.count, c.noted, c.issue])).toEqual([
["invented", 2, 1, true],
["unhappy_user", 1, 0, true],
["calm", 0, 2, false],
]);
expect(new Set(result.map((c) => c.key)).size).toBe(result.length);
expect(result[0].label).toBe("Invents answers");
expect(result[0].latest).toBe("latest");
expect(result[1].label).toBe("Unhappy user");
});
it("labels a check by its humanized id when the instruction is long", () => {
const long = "Agent takes a risky action (refund over limit, prod deploy) without required approval";
expect(checkLabel("no_approval", long)).toBe("No approval");
expect(checkLabel("expected_behavior", undefined)).toBe("Expected behavior");
expect(checkLabel("x", "Invents answers")).toBe("Invents answers");
});
it("is empty when nothing was flagged", () => {
expect(conclusions([review("a"), review("b")])).toEqual([]);
});
@ -124,7 +125,7 @@ describe("conclusions", () => {
it("filters traces to a group and lets everything through without one", () => {
const [invented] = conclusions([review("a", { verdicts: [issue("invented")] })]);
expect(inGroup(review("a", { verdicts: [issue("invented")] }), invented.key)).toBe(true);
expect(inGroup(review("b", { verdicts: [pattern("invented")] }), invented.key)).toBe(false);
expect(inGroup(review("b", { verdicts: [pattern("other")] }), invented.key)).toBe(false);
expect(inGroup(review("c"), null)).toBe(true);
});
@ -134,10 +135,28 @@ describe("conclusions", () => {
review("a", { verdicts: [issue("x"), pattern("y")] }),
review("b", { verdicts: [issue("x"), issue("z")] }),
]);
expect([...grownGroups(before, after)].sort()).toEqual(["issue:x", "issue:z"]);
expect([...grownGroups(before, after)].sort()).toEqual(["x", "z"]);
expect(grownGroups(after, after).size).toBe(0);
});
it("flashes a group that only gained a pattern trace", () => {
const before = conclusions([review("a", { verdicts: [pattern("y")] })]);
const after = conclusions([review("a", { verdicts: [pattern("y")] }), review("b", { verdicts: [pattern("y")] })]);
expect([...grownGroups(before, after)]).toEqual(["y"]);
});
it("lists upcoming traces above the one being read and finished ones below, newest first", () => {
const [a, b, c, d] = ["a", "b", "c", "d"].map((id) => review(id));
const rows = traceRows({ played: [a], current: b, pending: [c, d] }, 10);
expect(rows.map((row) => [row.review.execution_id, row.state])).toEqual([
["d", "queued"],
["c", "queued"],
["b", "reviewing"],
["a", "done"],
]);
expect(traceRows({ played: [a], current: b, pending: [c, d] }, 2)).toHaveLength(2);
});
it("gives a bar share bounded to the total", () => {
expect(share(3, 12)).toBe(0.25);
expect(share(5, 0)).toBe(0);
@ -349,19 +368,6 @@ describe("issue count", () => {
});
});
describe("drawer focus", () => {
const reviews = [review("a"), review("b")];
it("follows the live review until one is picked", () => {
expect(focusedReview(reviews, null, reviews[1])).toEqual({ review: reviews[1], following: true });
expect(focusedReview(reviews, reviewKey(reviews[0]), reviews[1])).toEqual({ review: reviews[0], following: false });
});
it("goes back to live when the picked review was dropped by the cap", () => {
expect(focusedReview(reviews, "gone@x", reviews[1])).toEqual({ review: reviews[1], following: true });
});
});
describe("incremental reviews", () => {
const at = (id: string, minute: number) => review(id, { at: `2026-10-03T16:${String(minute).padStart(2, "0")}:00Z` });

View file

@ -8,6 +8,7 @@ export interface Conclusion {
label: string;
latest: string;
count: number;
noted: number;
issue: boolean;
}
@ -52,10 +53,6 @@ export function outcome(review: Pick<Review, "cannot_assess" | "verdicts">): Out
return review.verdicts.some((v) => v.kind === "issue") ? "issue" : "clear";
}
export function tickerLine(review: Pick<Review, "agent" | "name" | "trace_id">): string {
return `reading ${review.agent || review.name} · ${review.trace_id.slice(0, 8)}`;
}
export function shortVerdict(review: Pick<Review, "cannot_assess" | "verdicts">): string {
const issue = review.verdicts.find((v) => v.kind === "issue");
if (issue) return issue.summary;
@ -100,22 +97,6 @@ export function issueCount(job: Pick<Job, "status" | "findings" | "reviews" | "r
return { count, scope: "" };
}
export function focusedReview(
reviews: readonly Review[],
pinned: string | null,
live: Review | null,
): { review: Review | null; following: boolean } {
const picked = pinned ? reviews.find((r) => reviewKey(r) === pinned) : undefined;
return picked ? { review: picked, following: false } : { review: live, following: true };
}
export function verdictLine(review: Pick<Review, "cannot_assess" | "verdicts">): string {
const issue = review.verdicts.find((v) => v.kind === "issue");
if (issue) return issue.summary;
if (review.cannot_assess) return "Not enough evidence to judge";
return review.verdicts[0]?.summary ?? "No issue observed";
}
const KEPT_REVIEWS = 200;
export interface ReviewFeed {
@ -152,45 +133,76 @@ export function playbackPhase(elapsed: number, duration: number, spans: number,
return { span, typed: Math.round(typing * chars), verdict: t >= READ_SHARE + TYPE_SHARE };
}
function checkLabel(checkId: string): string {
const SHORT_LABEL = 48;
function humanize(checkId: string): string {
const words = checkId.replace(/[_-]+/g, " ").trim();
return words ? words[0].toUpperCase() + words.slice(1) : checkId;
}
function groupKey(verdict: Pick<ReviewVerdict, "check_id" | "kind">): string {
return `${verdict.kind}:${verdict.check_id}`;
export function checkLabel(checkId: string, instruction: string | undefined): string {
return instruction && instruction.length <= SHORT_LABEL ? instruction : humanize(checkId);
}
function verdictsByCheck(review: Pick<Review, "verdicts">): Map<string, ReviewVerdict> {
const ranked = [...review.verdicts].sort((a, b) => Number(a.kind === "issue") - Number(b.kind === "issue"));
return new Map(ranked.map((verdict) => [verdict.check_id, verdict]));
}
export function conclusions(reviews: readonly Review[], checks: Settings["checks"] = []): Conclusion[] {
const instructions = new Map(checks.map((check) => [check.id, check.instruction]));
const grouped = reviews.reduce((groups, review) => {
const perTrace = new Map(review.verdicts.map((verdict) => [groupKey(verdict), verdict]));
return [...perTrace].reduce((next, [key, verdict]) => {
const prior = next.get(key);
return new Map(next).set(key, {
key,
return [...verdictsByCheck(review).values()].reduce((next, verdict) => {
const prior = next.get(verdict.check_id);
const issue = verdict.kind === "issue";
return new Map(next).set(verdict.check_id, {
key: verdict.check_id,
checkId: verdict.check_id,
label: instructions.get(verdict.check_id) ?? checkLabel(verdict.check_id),
latest: verdict.summary,
count: (prior?.count ?? 0) + 1,
issue: verdict.kind === "issue",
label: checkLabel(verdict.check_id, instructions.get(verdict.check_id)),
latest: issue || !prior ? verdict.summary : prior.latest,
count: (prior?.count ?? 0) + Number(issue),
noted: (prior?.noted ?? 0) + Number(!issue),
issue: (prior?.issue ?? false) || issue,
});
}, groups);
}, new Map<string, Conclusion>());
return [...grouped.values()].sort((a, b) => Number(b.issue) - Number(a.issue) || b.count - a.count);
return [...grouped.values()].sort((a, b) => b.count - a.count || b.noted - a.noted);
}
export function share(count: number, total: number): number {
return total > 0 ? Math.min(1, count / total) : 0;
}
export function inGroup(review: Pick<Review, "verdicts">, key: string | null): boolean {
return key === null || review.verdicts.some((verdict) => groupKey(verdict) === key);
export function inGroup(review: Pick<Review, "verdicts">, checkId: string | null): boolean {
return checkId === null || review.verdicts.some((verdict) => verdict.check_id === checkId);
}
const BRIEF_SENTENCES = 3;
export function briefReasoning(reasoning: string): string {
const sentences = reasoning.trim().split(/(?<=[.!?])\s+/);
return sentences.slice(0, BRIEF_SENTENCES).join(" ").trim();
}
export function grownGroups(before: readonly Conclusion[], after: readonly Conclusion[]): Set<string> {
const prior = new Map(before.map((group) => [group.key, group.count]));
return new Set(after.filter((group) => group.count > (prior.get(group.key) ?? 0)).map((group) => group.key));
const prior = new Map(before.map((group) => [group.key, group.count + group.noted]));
return new Set(
after.filter((group) => group.count + group.noted > (prior.get(group.key) ?? 0)).map((group) => group.key),
);
}
export type TraceRowState = "queued" | "reviewing" | "done";
export function traceRows(
playback: Pick<Playback, "played" | "current" | "pending">,
limit: number,
): { review: Review; state: TraceRowState }[] {
return [
...[...playback.pending].reverse().map((review) => ({ review, state: "queued" as const })),
...(playback.current ? [{ review: playback.current, state: "reviewing" as const }] : []),
...[...playback.played].reverse().map((review) => ({ review, state: "done" as const })),
].slice(0, limit);
}
export function decidedReviews(playback: Pick<Playback, "played" | "current">, verdictShown: boolean): Review[] {

View file

@ -1,126 +0,0 @@
import { describe, expect, it } from "vitest";
import { assistantReply, stepFailure, timeline, toolCall, userAsk } from "./spanPreview";
const OMIT = "\n[... preview omitted; read this span for evidence ...]\n";
const span = (kind: string, name: string, preview: string, cited = false) => ({
span_id: `${kind}-${name}`,
kind,
name,
preview,
cited,
});
const agentPreview =
'Input: [{"role": "system", "content": "You are a research assistant. Cite sources."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find a source for that."}]';
describe("user ask", () => {
it("takes the user's message and never the system prompt", () => {
expect(userAsk(agentPreview)).toBe("Find the median latency of the EU region.");
});
it("marks a user message cut off by the preview limit", () => {
const cut = 'Input: [{"role": "system", "content": "You are a billing agent."}, {"role": "user", "content": "Cancel C2473\'s subscript';
expect(userAsk(cut)).toBe("Cancel C2473's subscript…");
});
it("is null when only a system prompt is visible", () => {
expect(userAsk(`Input: [{"role": "system", "content": "You are a bill${OMIT}x"}]`)).toBeNull();
});
});
describe("assistant reply", () => {
it("reads the assistant output of an agent span", () => {
expect(assistantReply(agentPreview)).toBe("I could not find a source for that.");
});
it("recovers the end of a reply from a truncated llm span without its input", () => {
const llm = `Input: [{"role": "system", "content": "You are Acme's${OMIT}you want, I can also help you draft the message to your bank."}]\nStatus: STATUS_CODE_UNSET `;
expect(assistantReply(llm)).toBe("…you want, I can also help you draft the message to your bank.");
});
it("ignores a truncated tail that is a tool call or a JSON payload", () => {
const toolTail = `Input: [{"role": "system", "content": "You are a rese${OMIT}rch_docs", "arguments": "{\\"query\\":\\"latency\\"}"}}]}]\nStatus: STATUS_CODE_UNSET`;
expect(assistantReply(toolTail)).toBeNull();
});
it("is null when the output is empty", () => {
expect(assistantReply('Input: [{"role": "user", "content": "hi"}]\nOutput: \nStatus: STATUS_CODE_ERROR x')).toBeNull();
});
});
describe("tool call", () => {
it("names the tool and keeps args and result compact", () => {
const tool = span(
"tool",
"execute_tool lookup_order",
'Input: {"order_id":"A4160"}\nOutput: {"order_id": "A4160", "status": "processing"}\nStatus: STATUS_CODE_UNSET ',
);
expect(toolCall(tool)).toEqual({
kind: "tool",
name: "lookup_order",
args: '{"order_id":"A4160"}',
result: '{"order_id":"A4160","status":"processing"}',
error: false,
});
});
it("flags error status and error payloads but not ordinary results", () => {
const failed = 'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500';
expect(toolCall(span("tool", "execute_tool search_docs", failed)).error).toBe(true);
const denied = 'Input: {}\nOutput: {"status": "denied", "reason": "needs approval"}\nStatus: STATUS_CODE_UNSET';
expect(toolCall(span("tool", "execute_tool issue_refund", denied)).error).toBe(true);
const tests = 'Input: {}\nOutput: {"passed": 41, "failed": 3}\nStatus: STATUS_CODE_UNSET';
expect(toolCall(span("tool", "execute_tool run_tests", tests)).error).toBe(false);
});
it("reads a tool result that only survives in the preview tail", () => {
const cut = `Input: {"url":"https://example.com"}\nOutput: {"url": ${OMIT}, "text": "the figure is 37%"}\nStatus: STATUS_CODE_UNSET`;
expect(toolCall(span("tool", "execute_tool fetch_url", cut)).result).toBe('…, "text": "the figure is 37%"}');
});
it("keeps args clean when the preview cut lands inside the Output header", () => {
const cut = `Input: {"customer_id":"C3847"}\nO${OMIT}t month", "amount_usd": 416.67}\nStatus: STATUS_CODE_UNSET `;
const call = toolCall(span("tool", "execute_tool get_invoice", cut));
expect(call.args).toBe('{"customer_id":"C3847"}');
expect(call.result).toBe('…t month", "amount_usd": 416.67}');
});
});
describe("step failure", () => {
it("reports only non-ok statuses", () => {
expect(stepFailure("Output: \nStatus: STATUS_CODE_ERROR Request timed out.")).toBe("Request timed out.");
expect(stepFailure("Output: x\nStatus: STATUS_CODE_UNSET ")).toBeNull();
expect(stepFailure("plain")).toBeNull();
});
});
describe("timeline", () => {
it("reads like a conversation: ask, tool steps, then the final reply, without system text", () => {
const items = timeline([
span("agent", "invoke_agent research-agent", agentPreview),
span("llm", "chat gpt", `Input: [{"role": "system", "content": "You are a rese${OMIT}I could not find a source for that."}]`),
span(
"tool",
"execute_tool search_docs",
'Input: {"query":"EU latency"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500',
true,
),
]);
expect(items.map((i) => i.kind)).toEqual(["ask", "tool", "reply"]);
expect(items[0]).toMatchObject({ text: "Find the median latency of the EU region.", span: 0 });
expect(items[1]).toMatchObject({ name: "search_docs", error: true, span: 2 });
expect(items[2]).toMatchObject({ text: "I could not find a source for that.", span: 0 });
expect(JSON.stringify(items)).not.toContain("research assistant");
});
it("shows a failed model call as a failure step", () => {
const items = timeline([span("llm", "chat", "Input: x\nOutput: \nStatus: STATUS_CODE_ERROR Request timed out.")]);
expect(items).toEqual([{ kind: "failure", text: "Model call failed: Request timed out.", span: 0 }]);
});
it("falls back to short notes when nothing conversational can be recovered", () => {
const items = timeline([span("chain", "plan", 'Input: {"step": 1}')]);
expect(items).toEqual([{ kind: "note", label: "plan", text: '{"step":1}', span: 0 }]);
});
});

View file

@ -1,153 +0,0 @@
import { parseJson } from "@/components/view_logs/TraceView/traceUtils";
import type { Review } from "./types";
type Span = Review["spans"][number];
export type TimelineItem =
| { kind: "ask"; text: string; span: number }
| { kind: "tool"; name: string; args: string; result: string; error: boolean; span: number }
| { kind: "reply"; text: string; span: number }
| { kind: "failure"; text: string; span: number }
| { kind: "note"; label: string; text: string; span: number };
interface Message {
role: string;
content: string;
}
const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/;
const CUT_OUTPUT_HEADER = /\nO(?:u(?:t(?:p(?:u(?:t:?)?)?)?)?)? ?$/;
const SECTION_START =/(?:^|\n)(Input|Output|Status): ?/g;
const OK_STATUS = /^STATUS_CODE_(UNSET|OK)\b/;
const MESSAGE = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)(")?/g;
const TOOL_PROBLEM = /"error"|\bdenied\b|\bforbidden\b|\bunauthori[sz]ed\b|\bnot (?:allowed|permitted)\b/i;
const SNIPPET = 160;
function decode(escaped: string): string {
const parsed = parseJson(`"${escaped.replace(/\\u[0-9a-fA-F]{0,3}$|\\$/, "")}"`);
return typeof parsed === "string" ? parsed : escaped;
}
function tidy(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
function sections(text: string): Readonly<Partial<Record<"Input" | "Output" | "Status", string>>> {
const starts = [...text.matchAll(SECTION_START)];
return Object.fromEntries(
starts.map((match, n) => {
const from = (match.index ?? 0) + match[0].length;
const to = starts[n + 1]?.index ?? text.length;
return [match[1], text.slice(from, to)];
}),
);
}
function parts(preview: string) {
const [rawHead, tail = ""] = preview.split(OMITTED);
const head = rawHead.replace(CUT_OUTPUT_HEADER, "\nOutput: ");
const whole = sections(preview.replace(OMITTED, "\n"));
const ending = sections(tail);
const before = sections(head);
const cutOutput = before.Output !== undefined ? `…${tail.split(/\nStatus: /)[0] ?? ""}` : "";
return {
input: before.Input ?? "",
output: tail ? ending.Output ?? cutOutput : whole.Output ?? "",
status: (whole.Status ?? "").trim(),
tail: tail.split(/\nStatus: /)[0] ?? "",
truncated: !!tail,
};
}
function messages(text: string): Message[] {
return [...text.matchAll(MESSAGE)].map(([, role, content, closed]) => ({
role,
content: tidy(decode(content)) + (closed ? "" : "…"),
}));
}
function compact(text: string): string {
const trimmed = text.trim();
const parsed = parseJson(trimmed);
const flat = parsed !== null && typeof parsed === "object" ? JSON.stringify(parsed) : trimmed;
return tidy(flat).slice(0, SNIPPET);
}
export function stepFailure(preview: string): string | null {
const { status } = parts(preview);
if (!status || OK_STATUS.test(status)) return null;
return status.replace(/^STATUS_CODE_ERROR\s*/, "") || "failed";
}
export function userAsk(preview: string): string | null {
const asked = messages(parts(preview).input).filter((m) => m.role === "user" || m.role === "human");
return asked.at(-1)?.content || null;
}
function replyFromTail(tail: string): string | null {
const end = tail.trimEnd();
if (!end.endsWith('"}]') || end.includes('\\"') || /"function"|\{"|":\s/.test(end)) return null;
const text = tidy(end.slice(0, -3));
return text ? `…${text}` : null;
}
export function assistantReply(preview: string): string | null {
const { output, tail, truncated } = parts(preview);
const said = messages(output).filter((m) => m.role === "assistant" && m.content && m.content !== "…");
if (said.length) return said.at(-1)?.content ?? null;
return truncated && !output ? replyFromTail(tail) : null;
}
export function toolCall(span: Pick<Span, "name" | "preview">): Omit<Extract<TimelineItem, { kind: "tool" }>, "span"> {
const { input, output, status } = parts(span.preview);
const result = compact(output);
const failed = !!status && !OK_STATUS.test(status);
return {
kind: "tool",
name: span.name.replace(/^execute_tool\s+/, ""),
args: compact(input),
result: result || (failed ? stepFailure(span.preview) ?? "" : ""),
error: failed || TOOL_PROBLEM.test(output),
};
}
function overlaps(a: string, b: string): boolean {
const strip = (s: string) => s.replace(/…/g, "").trim();
const [x, y] = [strip(a), strip(b)];
return !!x && !!y && (x.includes(y) || y.includes(x));
}
function itemFor(span: Span, index: number): TimelineItem | null {
if (span.kind === "tool") return { ...toolCall(span), span: index };
if (span.kind === "llm") {
const failure = stepFailure(span.preview);
if (failure) return { kind: "failure", text: `Model call failed: ${failure}`, span: index };
const reply = assistantReply(span.preview);
return reply ? { kind: "reply", text: reply, span: index } : null;
}
return null;
}
function fallback(spans: readonly Span[]): TimelineItem[] {
return spans.map((span, index) => {
const { input, output } = parts(span.preview);
const said = messages(input).filter((m) => m.role !== "system").at(-1)?.content;
return { kind: "note", label: span.name, text: said ?? compact(output || input), span: index };
});
}
export function timeline(spans: readonly Span[]): TimelineItem[] {
const askAt = spans.findIndex((span) => userAsk(span.preview));
const ask = askAt >= 0 ? userAsk(spans[askAt].preview) : null;
const agentAt = spans.findIndex((span) => span.kind === "agent" && assistantReply(span.preview));
const final = agentAt >= 0 ? assistantReply(spans[agentAt].preview) : null;
const steps = spans.flatMap((span, index) => itemFor(span, index) ?? []);
const middle = final ? steps.filter((item) => item.kind !== "reply" || !overlaps(item.text, final)) : steps;
const items: TimelineItem[] = [
...(ask ? [{ kind: "ask" as const, text: ask, span: askAt }] : []),
...middle,
...(final ? [{ kind: "reply" as const, text: final, span: agentAt }] : []),
];
return items.length ? items : fallback(spans);
}