mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(lens): group live conclusions by check with short labels
There is now one group per check_id: issue traces are the main count and pattern traces a secondary note, so there are no duplicate red and grey cards. A long instruction falls back to the humanized check id. Adds briefReasoning and traceRows for the simplified trace list, and drops helpers nothing uses.
This commit is contained in:
parent
10260da169
commit
55d9425a82
6 changed files with 91 additions and 596 deletions
|
|
@ -1,178 +0,0 @@
|
|||
import { Wrench } from "lucide-react";
|
||||
|
||||
import { cn } from "@/lib/cva.config";
|
||||
|
||||
import { outcome, type Outcome, type Phase } from "../../model/live";
|
||||
import { timeline, type TimelineItem } from "../../model/spanPreview";
|
||||
import type { Review } from "../../model/types";
|
||||
import { ModelName } from "./LiveStrip";
|
||||
|
||||
const RED = "text-[#e5484d]";
|
||||
const CHIP: Record<Outcome, { label: string; className: string }> = {
|
||||
issue: { label: "Issue", className: "bg-[#e5484d]/10 text-[#e5484d] ring-1 ring-inset ring-[#e5484d]/30" },
|
||||
clear: { label: "Looks fine", className: "bg-muted text-foreground" },
|
||||
unknown: { label: "Not enough evidence", className: "bg-muted/60 text-muted-foreground" },
|
||||
};
|
||||
const PANE_LABEL = "text-[11px] font-medium text-muted-foreground";
|
||||
|
||||
function VerdictChip({ result, decided }: { result: Outcome; decided: boolean }) {
|
||||
if (!decided) {
|
||||
return <span className="rounded-full bg-muted/60 px-2 py-0.5 text-[11px] text-muted-foreground">Reading…</span>;
|
||||
}
|
||||
const chip = CHIP[result];
|
||||
return (
|
||||
<span
|
||||
data-testid="verdict-chip"
|
||||
className={cn(
|
||||
"rounded-full px-2 py-0.5 text-[11px] font-medium motion-safe:animate-in motion-safe:zoom-in-95 motion-safe:fade-in",
|
||||
chip.className,
|
||||
)}
|
||||
>
|
||||
{chip.label}
|
||||
</span>
|
||||
);
|
||||
}
|
||||
|
||||
function Bubble({ who, text, tone }: { who: string; text: string; tone: "ask" | "reply" }) {
|
||||
return (
|
||||
<div className={cn("flex flex-col gap-1", tone === "reply" && "items-end")}>
|
||||
<span className="text-[11px] text-muted-foreground">{who}</span>
|
||||
<p
|
||||
className={cn(
|
||||
"max-w-[92%] rounded-2xl px-3 py-2 text-[13px] leading-relaxed break-words whitespace-pre-wrap text-foreground",
|
||||
tone === "ask" ? "rounded-tl-sm bg-muted/70" : "rounded-tr-sm border bg-background",
|
||||
)}
|
||||
>
|
||||
{text}
|
||||
</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function ToolRow({ item }: { item: Extract<TimelineItem, { kind: "tool" }> }) {
|
||||
return (
|
||||
<div className="flex gap-2 text-[12px]">
|
||||
<Wrench aria-hidden="true" className={cn("mt-0.5 size-3.5 shrink-0 text-muted-foreground", item.error && RED)} />
|
||||
<div className="min-w-0 flex-1">
|
||||
<p className="flex min-w-0 gap-1.5">
|
||||
<span className="shrink-0 font-medium text-foreground">{item.name}</span>
|
||||
<span className="truncate text-muted-foreground">{item.args}</span>
|
||||
</p>
|
||||
{item.result && (
|
||||
<p className={cn("line-clamp-2 break-words", item.error ? RED : "text-muted-foreground")}>
|
||||
<span aria-hidden="true">↳ </span>
|
||||
{item.result}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function Step({ item }: { item: TimelineItem }) {
|
||||
switch (item.kind) {
|
||||
case "ask":
|
||||
return <Bubble who="User asked" text={item.text} tone="ask" />;
|
||||
case "reply":
|
||||
return <Bubble who="Agent replied" text={item.text} tone="reply" />;
|
||||
case "tool":
|
||||
return <ToolRow item={item} />;
|
||||
case "failure":
|
||||
return <p className={cn("text-[12px]", RED)}>{item.text}</p>;
|
||||
case "note":
|
||||
return (
|
||||
<p className="line-clamp-2 text-[12px] text-muted-foreground">
|
||||
<span className="mr-1.5 font-medium text-foreground">{item.label}</span>
|
||||
{item.text}
|
||||
</p>
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
function WhatHappened({ review, items, phase }: { review: Review; items: readonly TimelineItem[]; phase: Phase }) {
|
||||
if (!items.length) return <p className="text-[12px] text-muted-foreground">No trace parts were shown to the model</p>;
|
||||
return (
|
||||
<ol aria-label="What happened" className="flex flex-col gap-1">
|
||||
{items.map((item, index) => {
|
||||
const reading = phase.span === index;
|
||||
const cited = phase.verdict && !!review.spans[item.span]?.cited;
|
||||
return (
|
||||
<li
|
||||
key={`${item.kind}-${item.span}-${index}`}
|
||||
data-state={reading ? "reading" : cited ? "cited" : "idle"}
|
||||
className={cn(
|
||||
"rounded-lg px-2.5 py-2 transition-colors duration-150 motion-reduce:transition-none",
|
||||
reading && "bg-trace-row-hover ring-1 ring-inset ring-foreground/15",
|
||||
cited && "ring-[1.5px] ring-inset ring-[#e5484d]",
|
||||
)}
|
||||
>
|
||||
<Step item={item} />
|
||||
</li>
|
||||
);
|
||||
})}
|
||||
</ol>
|
||||
);
|
||||
}
|
||||
|
||||
export function ReadingPanel({ review, phase }: { review: Review; phase: Phase }) {
|
||||
const result = outcome(review);
|
||||
const items = timeline(review.spans);
|
||||
const typing = phase.typed < review.reasoning.length;
|
||||
const verdicts = review.verdicts.length
|
||||
? review.verdicts
|
||||
: [{ check_id: "", kind: "pattern" as const, summary: review.cannot_assess ? "Not enough evidence to judge" : "No issue observed" }];
|
||||
return (
|
||||
<div className="flex flex-col gap-5">
|
||||
<header className="flex items-start justify-between gap-3">
|
||||
<div className="min-w-0">
|
||||
<h3 className="truncate text-[14px] font-semibold text-foreground">{review.agent || review.name}</h3>
|
||||
<p className="truncate font-mono text-[11px] text-muted-foreground">trace {review.trace_id}</p>
|
||||
</div>
|
||||
<VerdictChip result={result} decided={phase.verdict} />
|
||||
</header>
|
||||
<section className="flex flex-col gap-2">
|
||||
<h4 className={PANE_LABEL}>What happened</h4>
|
||||
<WhatHappened review={review} items={items} phase={phase} />
|
||||
</section>
|
||||
<section aria-label="Reasoning" className="flex flex-col gap-2">
|
||||
<h4 className={cn(PANE_LABEL, "flex items-center gap-2")}>
|
||||
<span>Lens's reasoning</span>
|
||||
<ModelName model={review.model} />
|
||||
</h4>
|
||||
<p className="min-h-10 text-[13px] leading-relaxed whitespace-pre-wrap text-foreground">
|
||||
{review.reasoning ? review.reasoning.slice(0, phase.typed) : !typing && "No reasoning was returned."}
|
||||
{typing && <span aria-hidden="true" className="ml-px inline-block h-3.5 w-px translate-y-0.5 bg-foreground" />}
|
||||
</p>
|
||||
</section>
|
||||
<ul
|
||||
role="status"
|
||||
aria-label="Verdict"
|
||||
data-outcome={result}
|
||||
className={cn(
|
||||
"flex flex-col gap-1.5 transition-opacity duration-150 motion-reduce:transition-none",
|
||||
phase.verdict ? "opacity-100" : "opacity-0",
|
||||
)}
|
||||
>
|
||||
{phase.verdict &&
|
||||
verdicts.map((verdict, index) => (
|
||||
<li
|
||||
key={`${verdict.check_id}-${index}`}
|
||||
className={cn(
|
||||
"flex gap-2 text-[13px] leading-snug",
|
||||
verdict.kind === "issue" ? RED : result === "unknown" ? "text-muted-foreground" : "text-foreground",
|
||||
)}
|
||||
>
|
||||
<span
|
||||
aria-hidden="true"
|
||||
className={cn(
|
||||
"mt-1.5 size-1.5 shrink-0 rounded-full",
|
||||
verdict.kind === "issue" ? "bg-[#e5484d]" : "bg-muted-foreground/40",
|
||||
)}
|
||||
/>
|
||||
<span className="min-w-0">{verdict.summary}</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
|
@ -1,66 +0,0 @@
|
|||
import { Loader2 } from "lucide-react";
|
||||
|
||||
import { agoLabel } from "@/components/view_logs/TraceView/lensField";
|
||||
import { useNow } from "@/hooks/useNow";
|
||||
import { cn } from "@/lib/cva.config";
|
||||
|
||||
import { inGroup, outcome, queueRows, reviewKey, shortVerdict, type Playback } from "../../model/live";
|
||||
import type { Review } from "../../model/types";
|
||||
|
||||
const LIMIT = 60;
|
||||
|
||||
export function ReviewQueue({
|
||||
playback,
|
||||
live,
|
||||
focused,
|
||||
group,
|
||||
onPick,
|
||||
}: {
|
||||
playback: Pick<Playback, "played" | "current">;
|
||||
live: boolean;
|
||||
focused: Review | null;
|
||||
group: string | null;
|
||||
onPick: (review: Review) => void;
|
||||
}) {
|
||||
const now = useNow(5000);
|
||||
const rows = queueRows(playback, LIMIT).filter((review) => inGroup(review, group));
|
||||
if (!rows.length) return <p className="py-2 text-[12px] text-muted-foreground">No traces in this group yet.</p>;
|
||||
return (
|
||||
<ol aria-label="Reviewed traces" className="flex flex-col">
|
||||
{rows.map((review) => {
|
||||
const reading = live && review === playback.current;
|
||||
const result = outcome(review);
|
||||
const selected = review === focused;
|
||||
return (
|
||||
<li key={reviewKey(review)} className={reading ? "motion-safe:animate-in motion-safe:fade-in" : ""}>
|
||||
<button
|
||||
type="button"
|
||||
aria-current={selected ? "true" : undefined}
|
||||
onClick={() => onPick(review)}
|
||||
className={cn(
|
||||
"grid w-full grid-cols-[0.75rem_minmax(0,7.5rem)_minmax(0,1fr)_auto] items-center gap-2 rounded-md px-2 py-1.5 text-left text-[12px] hover:bg-muted",
|
||||
selected && "bg-background ring-[1.5px] ring-inset ring-foreground",
|
||||
)}
|
||||
>
|
||||
{reading ? (
|
||||
<Loader2 aria-label="Reviewing" className="size-3 text-muted-foreground motion-safe:animate-spin" />
|
||||
) : (
|
||||
<span
|
||||
aria-label={result}
|
||||
className={cn("size-1.5 rounded-full", result === "issue" ? "bg-[#e5484d]" : "bg-muted-foreground/40")}
|
||||
/>
|
||||
)}
|
||||
<span className="truncate text-foreground">{review.agent || review.name}</span>
|
||||
<span className={cn("truncate", result === "issue" ? "text-[#e5484d]" : "text-muted-foreground")}>
|
||||
{reading ? "reading…" : shortVerdict(review)}
|
||||
</span>
|
||||
<span className="text-[11px] tabular-nums text-muted-foreground">
|
||||
{agoLabel(Date.parse(review.at), now)}
|
||||
</span>
|
||||
</button>
|
||||
</li>
|
||||
);
|
||||
})}
|
||||
</ol>
|
||||
);
|
||||
}
|
||||
|
|
@ -1,9 +1,11 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
analysisModel,
|
||||
briefReasoning,
|
||||
checkLabel,
|
||||
conclusions,
|
||||
traceRows,
|
||||
decidedReviews,
|
||||
focusedReview,
|
||||
grownGroups,
|
||||
inGroup,
|
||||
share,
|
||||
|
|
@ -12,7 +14,6 @@ import {
|
|||
EMPTY_FEED,
|
||||
shortVerdict,
|
||||
stripState,
|
||||
tickerLine,
|
||||
liveJob,
|
||||
liveStats,
|
||||
outcome,
|
||||
|
|
@ -27,7 +28,6 @@ import {
|
|||
startPlayback,
|
||||
stepDuration,
|
||||
tokenLabel,
|
||||
verdictLine,
|
||||
} from "./live";
|
||||
import type { Job, Review } from "./types";
|
||||
|
||||
|
|
@ -83,40 +83,41 @@ describe("review outcome", () => {
|
|||
expect(outcome(review("a", { cannot_assess: true, verdicts: [issue("i")] }))).toBe("unknown");
|
||||
});
|
||||
|
||||
it("names the agent and short trace id in the ticker, falling back to the run name", () => {
|
||||
expect(tickerLine(review("a", { trace_id: "a91f3c02deadbeef" }))).toBe("reading support-bot · a91f3c02");
|
||||
expect(tickerLine(review("a", { agent: "", name: "refund", trace_id: "7d21" }))).toBe("reading refund · 7d21");
|
||||
});
|
||||
|
||||
it("leads the verdict line with the issue summary over patterns", () => {
|
||||
expect(verdictLine(review("a", { verdicts: [pattern("p", "fine"), issue("i", "made it up")] }))).toBe(
|
||||
"made it up",
|
||||
);
|
||||
expect(verdictLine(review("a"))).toBe("No issue observed");
|
||||
expect(verdictLine(review("a", { cannot_assess: true }))).toBe("Not enough evidence to judge");
|
||||
it("keeps the first few sentences of the reasoning", () => {
|
||||
expect(briefReasoning("One. Two? Three! Four. Five.")).toBe("One. Two? Three!");
|
||||
expect(briefReasoning("Saw 1.5 percent drop. Fine")).toBe("Saw 1.5 percent drop. Fine");
|
||||
expect(briefReasoning(" ")).toBe("");
|
||||
});
|
||||
});
|
||||
|
||||
describe("conclusions", () => {
|
||||
it("counts traces per check and kind, ranking issues before patterns, then by count", () => {
|
||||
it("makes one group per check, counting issue traces and noting pattern traces", () => {
|
||||
const reviews = [
|
||||
review("a", { verdicts: [pattern("calm"), issue("invented", "first")] }),
|
||||
review("b", { verdicts: [pattern("calm"), pattern("calm")] }),
|
||||
review("c", { verdicts: [issue("unhappy_user")] }),
|
||||
review("d", { verdicts: [issue("invented", "latest"), pattern("invented")] }),
|
||||
review("d", { verdicts: [issue("invented", "latest"), pattern("invented", "fine")] }),
|
||||
review("e", { verdicts: [pattern("invented", "fine")] }),
|
||||
];
|
||||
const result = conclusions(reviews, [{ id: "invented", instruction: "Invents answers", enabled: true }]);
|
||||
expect(result.map((c) => [c.checkId, c.count, c.issue])).toEqual([
|
||||
["invented", 2, true],
|
||||
["unhappy_user", 1, true],
|
||||
["calm", 2, false],
|
||||
["invented", 1, false],
|
||||
expect(result.map((c) => [c.checkId, c.count, c.noted, c.issue])).toEqual([
|
||||
["invented", 2, 1, true],
|
||||
["unhappy_user", 1, 0, true],
|
||||
["calm", 0, 2, false],
|
||||
]);
|
||||
expect(new Set(result.map((c) => c.key)).size).toBe(result.length);
|
||||
expect(result[0].label).toBe("Invents answers");
|
||||
expect(result[0].latest).toBe("latest");
|
||||
expect(result[1].label).toBe("Unhappy user");
|
||||
});
|
||||
|
||||
it("labels a check by its humanized id when the instruction is long", () => {
|
||||
const long = "Agent takes a risky action (refund over limit, prod deploy) without required approval";
|
||||
expect(checkLabel("no_approval", long)).toBe("No approval");
|
||||
expect(checkLabel("expected_behavior", undefined)).toBe("Expected behavior");
|
||||
expect(checkLabel("x", "Invents answers")).toBe("Invents answers");
|
||||
});
|
||||
|
||||
it("is empty when nothing was flagged", () => {
|
||||
expect(conclusions([review("a"), review("b")])).toEqual([]);
|
||||
});
|
||||
|
|
@ -124,7 +125,7 @@ describe("conclusions", () => {
|
|||
it("filters traces to a group and lets everything through without one", () => {
|
||||
const [invented] = conclusions([review("a", { verdicts: [issue("invented")] })]);
|
||||
expect(inGroup(review("a", { verdicts: [issue("invented")] }), invented.key)).toBe(true);
|
||||
expect(inGroup(review("b", { verdicts: [pattern("invented")] }), invented.key)).toBe(false);
|
||||
expect(inGroup(review("b", { verdicts: [pattern("other")] }), invented.key)).toBe(false);
|
||||
expect(inGroup(review("c"), null)).toBe(true);
|
||||
});
|
||||
|
||||
|
|
@ -134,10 +135,28 @@ describe("conclusions", () => {
|
|||
review("a", { verdicts: [issue("x"), pattern("y")] }),
|
||||
review("b", { verdicts: [issue("x"), issue("z")] }),
|
||||
]);
|
||||
expect([...grownGroups(before, after)].sort()).toEqual(["issue:x", "issue:z"]);
|
||||
expect([...grownGroups(before, after)].sort()).toEqual(["x", "z"]);
|
||||
expect(grownGroups(after, after).size).toBe(0);
|
||||
});
|
||||
|
||||
it("flashes a group that only gained a pattern trace", () => {
|
||||
const before = conclusions([review("a", { verdicts: [pattern("y")] })]);
|
||||
const after = conclusions([review("a", { verdicts: [pattern("y")] }), review("b", { verdicts: [pattern("y")] })]);
|
||||
expect([...grownGroups(before, after)]).toEqual(["y"]);
|
||||
});
|
||||
|
||||
it("lists upcoming traces above the one being read and finished ones below, newest first", () => {
|
||||
const [a, b, c, d] = ["a", "b", "c", "d"].map((id) => review(id));
|
||||
const rows = traceRows({ played: [a], current: b, pending: [c, d] }, 10);
|
||||
expect(rows.map((row) => [row.review.execution_id, row.state])).toEqual([
|
||||
["d", "queued"],
|
||||
["c", "queued"],
|
||||
["b", "reviewing"],
|
||||
["a", "done"],
|
||||
]);
|
||||
expect(traceRows({ played: [a], current: b, pending: [c, d] }, 2)).toHaveLength(2);
|
||||
});
|
||||
|
||||
it("gives a bar share bounded to the total", () => {
|
||||
expect(share(3, 12)).toBe(0.25);
|
||||
expect(share(5, 0)).toBe(0);
|
||||
|
|
@ -349,19 +368,6 @@ describe("issue count", () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe("drawer focus", () => {
|
||||
const reviews = [review("a"), review("b")];
|
||||
|
||||
it("follows the live review until one is picked", () => {
|
||||
expect(focusedReview(reviews, null, reviews[1])).toEqual({ review: reviews[1], following: true });
|
||||
expect(focusedReview(reviews, reviewKey(reviews[0]), reviews[1])).toEqual({ review: reviews[0], following: false });
|
||||
});
|
||||
|
||||
it("goes back to live when the picked review was dropped by the cap", () => {
|
||||
expect(focusedReview(reviews, "gone@x", reviews[1])).toEqual({ review: reviews[1], following: true });
|
||||
});
|
||||
});
|
||||
|
||||
describe("incremental reviews", () => {
|
||||
const at = (id: string, minute: number) => review(id, { at: `2026-10-03T16:${String(minute).padStart(2, "0")}:00Z` });
|
||||
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ export interface Conclusion {
|
|||
label: string;
|
||||
latest: string;
|
||||
count: number;
|
||||
noted: number;
|
||||
issue: boolean;
|
||||
}
|
||||
|
||||
|
|
@ -52,10 +53,6 @@ export function outcome(review: Pick<Review, "cannot_assess" | "verdicts">): Out
|
|||
return review.verdicts.some((v) => v.kind === "issue") ? "issue" : "clear";
|
||||
}
|
||||
|
||||
export function tickerLine(review: Pick<Review, "agent" | "name" | "trace_id">): string {
|
||||
return `reading ${review.agent || review.name} · ${review.trace_id.slice(0, 8)}`;
|
||||
}
|
||||
|
||||
export function shortVerdict(review: Pick<Review, "cannot_assess" | "verdicts">): string {
|
||||
const issue = review.verdicts.find((v) => v.kind === "issue");
|
||||
if (issue) return issue.summary;
|
||||
|
|
@ -100,22 +97,6 @@ export function issueCount(job: Pick<Job, "status" | "findings" | "reviews" | "r
|
|||
return { count, scope: "" };
|
||||
}
|
||||
|
||||
export function focusedReview(
|
||||
reviews: readonly Review[],
|
||||
pinned: string | null,
|
||||
live: Review | null,
|
||||
): { review: Review | null; following: boolean } {
|
||||
const picked = pinned ? reviews.find((r) => reviewKey(r) === pinned) : undefined;
|
||||
return picked ? { review: picked, following: false } : { review: live, following: true };
|
||||
}
|
||||
|
||||
export function verdictLine(review: Pick<Review, "cannot_assess" | "verdicts">): string {
|
||||
const issue = review.verdicts.find((v) => v.kind === "issue");
|
||||
if (issue) return issue.summary;
|
||||
if (review.cannot_assess) return "Not enough evidence to judge";
|
||||
return review.verdicts[0]?.summary ?? "No issue observed";
|
||||
}
|
||||
|
||||
const KEPT_REVIEWS = 200;
|
||||
|
||||
export interface ReviewFeed {
|
||||
|
|
@ -152,45 +133,76 @@ export function playbackPhase(elapsed: number, duration: number, spans: number,
|
|||
return { span, typed: Math.round(typing * chars), verdict: t >= READ_SHARE + TYPE_SHARE };
|
||||
}
|
||||
|
||||
function checkLabel(checkId: string): string {
|
||||
const SHORT_LABEL = 48;
|
||||
|
||||
function humanize(checkId: string): string {
|
||||
const words = checkId.replace(/[_-]+/g, " ").trim();
|
||||
return words ? words[0].toUpperCase() + words.slice(1) : checkId;
|
||||
}
|
||||
|
||||
function groupKey(verdict: Pick<ReviewVerdict, "check_id" | "kind">): string {
|
||||
return `${verdict.kind}:${verdict.check_id}`;
|
||||
export function checkLabel(checkId: string, instruction: string | undefined): string {
|
||||
return instruction && instruction.length <= SHORT_LABEL ? instruction : humanize(checkId);
|
||||
}
|
||||
|
||||
function verdictsByCheck(review: Pick<Review, "verdicts">): Map<string, ReviewVerdict> {
|
||||
const ranked = [...review.verdicts].sort((a, b) => Number(a.kind === "issue") - Number(b.kind === "issue"));
|
||||
return new Map(ranked.map((verdict) => [verdict.check_id, verdict]));
|
||||
}
|
||||
|
||||
export function conclusions(reviews: readonly Review[], checks: Settings["checks"] = []): Conclusion[] {
|
||||
const instructions = new Map(checks.map((check) => [check.id, check.instruction]));
|
||||
const grouped = reviews.reduce((groups, review) => {
|
||||
const perTrace = new Map(review.verdicts.map((verdict) => [groupKey(verdict), verdict]));
|
||||
return [...perTrace].reduce((next, [key, verdict]) => {
|
||||
const prior = next.get(key);
|
||||
return new Map(next).set(key, {
|
||||
key,
|
||||
return [...verdictsByCheck(review).values()].reduce((next, verdict) => {
|
||||
const prior = next.get(verdict.check_id);
|
||||
const issue = verdict.kind === "issue";
|
||||
return new Map(next).set(verdict.check_id, {
|
||||
key: verdict.check_id,
|
||||
checkId: verdict.check_id,
|
||||
label: instructions.get(verdict.check_id) ?? checkLabel(verdict.check_id),
|
||||
latest: verdict.summary,
|
||||
count: (prior?.count ?? 0) + 1,
|
||||
issue: verdict.kind === "issue",
|
||||
label: checkLabel(verdict.check_id, instructions.get(verdict.check_id)),
|
||||
latest: issue || !prior ? verdict.summary : prior.latest,
|
||||
count: (prior?.count ?? 0) + Number(issue),
|
||||
noted: (prior?.noted ?? 0) + Number(!issue),
|
||||
issue: (prior?.issue ?? false) || issue,
|
||||
});
|
||||
}, groups);
|
||||
}, new Map<string, Conclusion>());
|
||||
return [...grouped.values()].sort((a, b) => Number(b.issue) - Number(a.issue) || b.count - a.count);
|
||||
return [...grouped.values()].sort((a, b) => b.count - a.count || b.noted - a.noted);
|
||||
}
|
||||
|
||||
export function share(count: number, total: number): number {
|
||||
return total > 0 ? Math.min(1, count / total) : 0;
|
||||
}
|
||||
|
||||
export function inGroup(review: Pick<Review, "verdicts">, key: string | null): boolean {
|
||||
return key === null || review.verdicts.some((verdict) => groupKey(verdict) === key);
|
||||
export function inGroup(review: Pick<Review, "verdicts">, checkId: string | null): boolean {
|
||||
return checkId === null || review.verdicts.some((verdict) => verdict.check_id === checkId);
|
||||
}
|
||||
|
||||
const BRIEF_SENTENCES = 3;
|
||||
|
||||
export function briefReasoning(reasoning: string): string {
|
||||
const sentences = reasoning.trim().split(/(?<=[.!?])\s+/);
|
||||
return sentences.slice(0, BRIEF_SENTENCES).join(" ").trim();
|
||||
}
|
||||
|
||||
|
||||
export function grownGroups(before: readonly Conclusion[], after: readonly Conclusion[]): Set<string> {
|
||||
const prior = new Map(before.map((group) => [group.key, group.count]));
|
||||
return new Set(after.filter((group) => group.count > (prior.get(group.key) ?? 0)).map((group) => group.key));
|
||||
const prior = new Map(before.map((group) => [group.key, group.count + group.noted]));
|
||||
return new Set(
|
||||
after.filter((group) => group.count + group.noted > (prior.get(group.key) ?? 0)).map((group) => group.key),
|
||||
);
|
||||
}
|
||||
|
||||
export type TraceRowState = "queued" | "reviewing" | "done";
|
||||
|
||||
export function traceRows(
|
||||
playback: Pick<Playback, "played" | "current" | "pending">,
|
||||
limit: number,
|
||||
): { review: Review; state: TraceRowState }[] {
|
||||
return [
|
||||
...[...playback.pending].reverse().map((review) => ({ review, state: "queued" as const })),
|
||||
...(playback.current ? [{ review: playback.current, state: "reviewing" as const }] : []),
|
||||
...[...playback.played].reverse().map((review) => ({ review, state: "done" as const })),
|
||||
].slice(0, limit);
|
||||
}
|
||||
|
||||
export function decidedReviews(playback: Pick<Playback, "played" | "current">, verdictShown: boolean): Review[] {
|
||||
|
|
|
|||
|
|
@ -1,126 +0,0 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { assistantReply, stepFailure, timeline, toolCall, userAsk } from "./spanPreview";
|
||||
|
||||
const OMIT = "\n[... preview omitted; read this span for evidence ...]\n";
|
||||
|
||||
const span = (kind: string, name: string, preview: string, cited = false) => ({
|
||||
span_id: `${kind}-${name}`,
|
||||
kind,
|
||||
name,
|
||||
preview,
|
||||
cited,
|
||||
});
|
||||
|
||||
const agentPreview =
|
||||
'Input: [{"role": "system", "content": "You are a research assistant. Cite sources."}, {"role": "user", "content": "Find the median latency of the EU region."}]\nOutput: [{"role": "assistant", "content": "I could not find a source for that."}]';
|
||||
|
||||
describe("user ask", () => {
|
||||
it("takes the user's message and never the system prompt", () => {
|
||||
expect(userAsk(agentPreview)).toBe("Find the median latency of the EU region.");
|
||||
});
|
||||
|
||||
it("marks a user message cut off by the preview limit", () => {
|
||||
const cut = 'Input: [{"role": "system", "content": "You are a billing agent."}, {"role": "user", "content": "Cancel C2473\'s subscript';
|
||||
expect(userAsk(cut)).toBe("Cancel C2473's subscript…");
|
||||
});
|
||||
|
||||
it("is null when only a system prompt is visible", () => {
|
||||
expect(userAsk(`Input: [{"role": "system", "content": "You are a bill${OMIT}x"}]`)).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("assistant reply", () => {
|
||||
it("reads the assistant output of an agent span", () => {
|
||||
expect(assistantReply(agentPreview)).toBe("I could not find a source for that.");
|
||||
});
|
||||
|
||||
it("recovers the end of a reply from a truncated llm span without its input", () => {
|
||||
const llm = `Input: [{"role": "system", "content": "You are Acme's${OMIT}you want, I can also help you draft the message to your bank."}]\nStatus: STATUS_CODE_UNSET `;
|
||||
expect(assistantReply(llm)).toBe("…you want, I can also help you draft the message to your bank.");
|
||||
});
|
||||
|
||||
it("ignores a truncated tail that is a tool call or a JSON payload", () => {
|
||||
const toolTail = `Input: [{"role": "system", "content": "You are a rese${OMIT}rch_docs", "arguments": "{\\"query\\":\\"latency\\"}"}}]}]\nStatus: STATUS_CODE_UNSET`;
|
||||
expect(assistantReply(toolTail)).toBeNull();
|
||||
});
|
||||
|
||||
it("is null when the output is empty", () => {
|
||||
expect(assistantReply('Input: [{"role": "user", "content": "hi"}]\nOutput: \nStatus: STATUS_CODE_ERROR x')).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("tool call", () => {
|
||||
it("names the tool and keeps args and result compact", () => {
|
||||
const tool = span(
|
||||
"tool",
|
||||
"execute_tool lookup_order",
|
||||
'Input: {"order_id":"A4160"}\nOutput: {"order_id": "A4160", "status": "processing"}\nStatus: STATUS_CODE_UNSET ',
|
||||
);
|
||||
expect(toolCall(tool)).toEqual({
|
||||
kind: "tool",
|
||||
name: "lookup_order",
|
||||
args: '{"order_id":"A4160"}',
|
||||
result: '{"order_id":"A4160","status":"processing"}',
|
||||
error: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("flags error status and error payloads but not ordinary results", () => {
|
||||
const failed = 'Input: {"query":"q"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500';
|
||||
expect(toolCall(span("tool", "execute_tool search_docs", failed)).error).toBe(true);
|
||||
const denied = 'Input: {}\nOutput: {"status": "denied", "reason": "needs approval"}\nStatus: STATUS_CODE_UNSET';
|
||||
expect(toolCall(span("tool", "execute_tool issue_refund", denied)).error).toBe(true);
|
||||
const tests = 'Input: {}\nOutput: {"passed": 41, "failed": 3}\nStatus: STATUS_CODE_UNSET';
|
||||
expect(toolCall(span("tool", "execute_tool run_tests", tests)).error).toBe(false);
|
||||
});
|
||||
|
||||
it("reads a tool result that only survives in the preview tail", () => {
|
||||
const cut = `Input: {"url":"https://example.com"}\nOutput: {"url": ${OMIT}, "text": "the figure is 37%"}\nStatus: STATUS_CODE_UNSET`;
|
||||
expect(toolCall(span("tool", "execute_tool fetch_url", cut)).result).toBe('…, "text": "the figure is 37%"}');
|
||||
});
|
||||
|
||||
it("keeps args clean when the preview cut lands inside the Output header", () => {
|
||||
const cut = `Input: {"customer_id":"C3847"}\nO${OMIT}t month", "amount_usd": 416.67}\nStatus: STATUS_CODE_UNSET `;
|
||||
const call = toolCall(span("tool", "execute_tool get_invoice", cut));
|
||||
expect(call.args).toBe('{"customer_id":"C3847"}');
|
||||
expect(call.result).toBe('…t month", "amount_usd": 416.67}');
|
||||
});
|
||||
});
|
||||
|
||||
describe("step failure", () => {
|
||||
it("reports only non-ok statuses", () => {
|
||||
expect(stepFailure("Output: \nStatus: STATUS_CODE_ERROR Request timed out.")).toBe("Request timed out.");
|
||||
expect(stepFailure("Output: x\nStatus: STATUS_CODE_UNSET ")).toBeNull();
|
||||
expect(stepFailure("plain")).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("timeline", () => {
|
||||
it("reads like a conversation: ask, tool steps, then the final reply, without system text", () => {
|
||||
const items = timeline([
|
||||
span("agent", "invoke_agent research-agent", agentPreview),
|
||||
span("llm", "chat gpt", `Input: [{"role": "system", "content": "You are a rese${OMIT}I could not find a source for that."}]`),
|
||||
span(
|
||||
"tool",
|
||||
"execute_tool search_docs",
|
||||
'Input: {"query":"EU latency"}\nOutput: {"error": "500 search index unavailable"}\nStatus: STATUS_CODE_ERROR 500',
|
||||
true,
|
||||
),
|
||||
]);
|
||||
expect(items.map((i) => i.kind)).toEqual(["ask", "tool", "reply"]);
|
||||
expect(items[0]).toMatchObject({ text: "Find the median latency of the EU region.", span: 0 });
|
||||
expect(items[1]).toMatchObject({ name: "search_docs", error: true, span: 2 });
|
||||
expect(items[2]).toMatchObject({ text: "I could not find a source for that.", span: 0 });
|
||||
expect(JSON.stringify(items)).not.toContain("research assistant");
|
||||
});
|
||||
|
||||
it("shows a failed model call as a failure step", () => {
|
||||
const items = timeline([span("llm", "chat", "Input: x\nOutput: \nStatus: STATUS_CODE_ERROR Request timed out.")]);
|
||||
expect(items).toEqual([{ kind: "failure", text: "Model call failed: Request timed out.", span: 0 }]);
|
||||
});
|
||||
|
||||
it("falls back to short notes when nothing conversational can be recovered", () => {
|
||||
const items = timeline([span("chain", "plan", 'Input: {"step": 1}')]);
|
||||
expect(items).toEqual([{ kind: "note", label: "plan", text: '{"step":1}', span: 0 }]);
|
||||
});
|
||||
});
|
||||
|
|
@ -1,153 +0,0 @@
|
|||
import { parseJson } from "@/components/view_logs/TraceView/traceUtils";
|
||||
|
||||
import type { Review } from "./types";
|
||||
|
||||
type Span = Review["spans"][number];
|
||||
|
||||
export type TimelineItem =
|
||||
| { kind: "ask"; text: string; span: number }
|
||||
| { kind: "tool"; name: string; args: string; result: string; error: boolean; span: number }
|
||||
| { kind: "reply"; text: string; span: number }
|
||||
| { kind: "failure"; text: string; span: number }
|
||||
| { kind: "note"; label: string; text: string; span: number };
|
||||
|
||||
interface Message {
|
||||
role: string;
|
||||
content: string;
|
||||
}
|
||||
|
||||
const OMITTED = /\n?\[\.\.\. preview omitted; read this span for evidence \.\.\.\]\n?/;
|
||||
const CUT_OUTPUT_HEADER = /\nO(?:u(?:t(?:p(?:u(?:t:?)?)?)?)?)? ?$/;
|
||||
const SECTION_START =/(?:^|\n)(Input|Output|Status): ?/g;
|
||||
const OK_STATUS = /^STATUS_CODE_(UNSET|OK)\b/;
|
||||
const MESSAGE = /"role":\s*"(\w+)",\s*"content":\s*"((?:[^"\\]|\\.)*)(")?/g;
|
||||
const TOOL_PROBLEM = /"error"|\bdenied\b|\bforbidden\b|\bunauthori[sz]ed\b|\bnot (?:allowed|permitted)\b/i;
|
||||
const SNIPPET = 160;
|
||||
|
||||
function decode(escaped: string): string {
|
||||
const parsed = parseJson(`"${escaped.replace(/\\u[0-9a-fA-F]{0,3}$|\\$/, "")}"`);
|
||||
return typeof parsed === "string" ? parsed : escaped;
|
||||
}
|
||||
|
||||
function tidy(text: string): string {
|
||||
return text.replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
function sections(text: string): Readonly<Partial<Record<"Input" | "Output" | "Status", string>>> {
|
||||
const starts = [...text.matchAll(SECTION_START)];
|
||||
return Object.fromEntries(
|
||||
starts.map((match, n) => {
|
||||
const from = (match.index ?? 0) + match[0].length;
|
||||
const to = starts[n + 1]?.index ?? text.length;
|
||||
return [match[1], text.slice(from, to)];
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
function parts(preview: string) {
|
||||
const [rawHead, tail = ""] = preview.split(OMITTED);
|
||||
const head = rawHead.replace(CUT_OUTPUT_HEADER, "\nOutput: ");
|
||||
const whole = sections(preview.replace(OMITTED, "\n"));
|
||||
const ending = sections(tail);
|
||||
const before = sections(head);
|
||||
const cutOutput = before.Output !== undefined ? `…${tail.split(/\nStatus: /)[0] ?? ""}` : "";
|
||||
return {
|
||||
input: before.Input ?? "",
|
||||
output: tail ? ending.Output ?? cutOutput : whole.Output ?? "",
|
||||
status: (whole.Status ?? "").trim(),
|
||||
tail: tail.split(/\nStatus: /)[0] ?? "",
|
||||
truncated: !!tail,
|
||||
};
|
||||
}
|
||||
|
||||
function messages(text: string): Message[] {
|
||||
return [...text.matchAll(MESSAGE)].map(([, role, content, closed]) => ({
|
||||
role,
|
||||
content: tidy(decode(content)) + (closed ? "" : "…"),
|
||||
}));
|
||||
}
|
||||
|
||||
function compact(text: string): string {
|
||||
const trimmed = text.trim();
|
||||
const parsed = parseJson(trimmed);
|
||||
const flat = parsed !== null && typeof parsed === "object" ? JSON.stringify(parsed) : trimmed;
|
||||
return tidy(flat).slice(0, SNIPPET);
|
||||
}
|
||||
|
||||
export function stepFailure(preview: string): string | null {
|
||||
const { status } = parts(preview);
|
||||
if (!status || OK_STATUS.test(status)) return null;
|
||||
return status.replace(/^STATUS_CODE_ERROR\s*/, "") || "failed";
|
||||
}
|
||||
|
||||
export function userAsk(preview: string): string | null {
|
||||
const asked = messages(parts(preview).input).filter((m) => m.role === "user" || m.role === "human");
|
||||
return asked.at(-1)?.content || null;
|
||||
}
|
||||
|
||||
function replyFromTail(tail: string): string | null {
|
||||
const end = tail.trimEnd();
|
||||
if (!end.endsWith('"}]') || end.includes('\\"') || /"function"|\{"|":\s/.test(end)) return null;
|
||||
const text = tidy(end.slice(0, -3));
|
||||
return text ? `…${text}` : null;
|
||||
}
|
||||
|
||||
export function assistantReply(preview: string): string | null {
|
||||
const { output, tail, truncated } = parts(preview);
|
||||
const said = messages(output).filter((m) => m.role === "assistant" && m.content && m.content !== "…");
|
||||
if (said.length) return said.at(-1)?.content ?? null;
|
||||
return truncated && !output ? replyFromTail(tail) : null;
|
||||
}
|
||||
|
||||
export function toolCall(span: Pick<Span, "name" | "preview">): Omit<Extract<TimelineItem, { kind: "tool" }>, "span"> {
|
||||
const { input, output, status } = parts(span.preview);
|
||||
const result = compact(output);
|
||||
const failed = !!status && !OK_STATUS.test(status);
|
||||
return {
|
||||
kind: "tool",
|
||||
name: span.name.replace(/^execute_tool\s+/, ""),
|
||||
args: compact(input),
|
||||
result: result || (failed ? stepFailure(span.preview) ?? "" : ""),
|
||||
error: failed || TOOL_PROBLEM.test(output),
|
||||
};
|
||||
}
|
||||
|
||||
function overlaps(a: string, b: string): boolean {
|
||||
const strip = (s: string) => s.replace(/…/g, "").trim();
|
||||
const [x, y] = [strip(a), strip(b)];
|
||||
return !!x && !!y && (x.includes(y) || y.includes(x));
|
||||
}
|
||||
|
||||
function itemFor(span: Span, index: number): TimelineItem | null {
|
||||
if (span.kind === "tool") return { ...toolCall(span), span: index };
|
||||
if (span.kind === "llm") {
|
||||
const failure = stepFailure(span.preview);
|
||||
if (failure) return { kind: "failure", text: `Model call failed: ${failure}`, span: index };
|
||||
const reply = assistantReply(span.preview);
|
||||
return reply ? { kind: "reply", text: reply, span: index } : null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function fallback(spans: readonly Span[]): TimelineItem[] {
|
||||
return spans.map((span, index) => {
|
||||
const { input, output } = parts(span.preview);
|
||||
const said = messages(input).filter((m) => m.role !== "system").at(-1)?.content;
|
||||
return { kind: "note", label: span.name, text: said ?? compact(output || input), span: index };
|
||||
});
|
||||
}
|
||||
|
||||
export function timeline(spans: readonly Span[]): TimelineItem[] {
|
||||
const askAt = spans.findIndex((span) => userAsk(span.preview));
|
||||
const ask = askAt >= 0 ? userAsk(spans[askAt].preview) : null;
|
||||
const agentAt = spans.findIndex((span) => span.kind === "agent" && assistantReply(span.preview));
|
||||
const final = agentAt >= 0 ? assistantReply(spans[agentAt].preview) : null;
|
||||
const steps = spans.flatMap((span, index) => itemFor(span, index) ?? []);
|
||||
const middle = final ? steps.filter((item) => item.kind !== "reply" || !overlaps(item.text, final)) : steps;
|
||||
const items: TimelineItem[] = [
|
||||
...(ask ? [{ kind: "ask" as const, text: ask, span: askAt }] : []),
|
||||
...middle,
|
||||
...(final ? [{ kind: "reply" as const, text: final, span: agentAt }] : []),
|
||||
];
|
||||
return items.length ? items : fallback(spans);
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue